{"timestamp_utc": "2026-04-11T22:36:55Z", "mode": "train", "global_step": 651, "epoch": 0.026147728642005062, "loss": -0.0071, "grad_norm": 9.100071907043457, "learning_rate": 8.03030303030303e-06, "num_tokens": 1409555.0, "completions/mean_length": 35.125, "completions/min_length": 33.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.125, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9935092329978943, "rewards/meter/std": 0.000985646271146834, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9935092329978943, "rewards/total_composite/std": 0.000985646271146834, "reward": 0.9935092329978943, "reward_std": 0.0009856420801952481, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04576735198497772, "sampling/sampling_logp_difference/max": 0.9458228349685669, "sampling/importance_sampling_ratio/min": 0.3883599042892456, "sampling/importance_sampling_ratio/mean": 1.011391282081604, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17123969458043575, "clip_ratio/low_mean": 0.028315248200669885, "clip_ratio/low_min": 0.028315248200669885, "clip_ratio/high_mean": 0.010521235642954707, "clip_ratio/high_max": 0.010521235642954707, "clip_ratio/region_mean": 0.03883648384362459, "reward_total_mean": 0.9935092329978943, "reward_meter_mean": 0.9935092329978943, "reward_meter_std": 0.000985646271146834, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9935092329978943, "reward_total_composite_std": 0.000985646271146834} {"timestamp_utc": "2026-04-11T22:37:05Z", "mode": "train", "global_step": 652, "epoch": 0.026187894123790016, "loss": -0.2894, "grad_norm": 0.4141400456428528, "learning_rate": 8.027272727272728e-06, "num_tokens": 1413749.0, "completions/mean_length": 374.25, "completions/min_length": 343.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 354.5714416503906, "completions/min_terminated_length": 343.0, "completions/max_terminated_length": 363.0, "rewards/meter/mean": 0.8732885122299194, "rewards/meter/std": 0.3528631627559662, "rewards/count_adherence/mean": 0.824999988079071, "rewards/count_adherence/std": 0.337003618478775, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.17251461744308472, "rewards/repeat_penalty/std": 0.33435773849487305, "rewards/total_composite/mean": 0.04464861750602722, "rewards/total_composite/std": 0.018085261806845665, "reward": 0.04464861750602722, "reward_std": 0.018085261806845665, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004473666660487652, "sampling/sampling_logp_difference/max": 0.7091238498687744, "sampling/importance_sampling_ratio/min": 0.4920751452445984, "sampling/importance_sampling_ratio/mean": 1.0013567209243774, "sampling/importance_sampling_ratio/max": 1.834702491760254, "entropy": 0.019337893230840564, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0035446555411908776, "clip_ratio/high_max": 0.0035446555411908776, "clip_ratio/region_mean": 0.0035446555411908776, "reward_total_mean": 0.04464861750602722, "reward_meter_mean": 0.8732885122299194, "reward_meter_std": 0.3528631627559662, "reward_count_adherence_mean": 0.824999988079071, "reward_count_adherence_std": 0.337003618478775, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.17251461744308472, "reward_repeat_penalty_std": 0.33435773849487305, "reward_total_composite_mean": 0.04464861750602722, "reward_total_composite_std": 0.018085261806845665} {"timestamp_utc": "2026-04-11T22:37:15Z", "mode": "train", "global_step": 653, "epoch": 0.02622805960557497, "loss": 0.1416, "grad_norm": 1.191830039024353, "learning_rate": 8.024242424242425e-06, "num_tokens": 1415617.0, "completions/mean_length": 133.5, "completions/min_length": 78.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 79.42857360839844, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.9687988758087158, "rewards/meter/std": 0.06074121594429016, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4583333432674408, "rewards/repeat_penalty/std": 0.24800792336463928, "rewards/total_composite/mean": 0.43250638246536255, "rewards/total_composite/std": 0.1944577693939209, "reward": 0.43250638246536255, "reward_std": 0.1944577544927597, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014217361807823181, "sampling/sampling_logp_difference/max": 0.9473090171813965, "sampling/importance_sampling_ratio/min": 0.3877831697463989, "sampling/importance_sampling_ratio/mean": 1.0012526512145996, "sampling/importance_sampling_ratio/max": 1.5146371126174927, "entropy": 0.0790142323821783, "clip_ratio/low_mean": 0.007873826543800533, "clip_ratio/low_min": 0.007873826543800533, "clip_ratio/high_mean": 0.0029761905316263437, "clip_ratio/high_max": 0.0029761905316263437, "clip_ratio/region_mean": 0.010850017075426877, "reward_total_mean": 0.43250638246536255, "reward_meter_mean": 0.9687988758087158, "reward_meter_std": 0.06074121594429016, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4583333432674408, "reward_repeat_penalty_std": 0.24800792336463928, "reward_total_composite_mean": 0.43250638246536255, "reward_total_composite_std": 0.1944577693939209} {"timestamp_utc": "2026-04-11T22:37:21Z", "mode": "train", "global_step": 654, "epoch": 0.026268225087359924, "loss": -0.0232, "grad_norm": 1.927111268043518, "learning_rate": 8.021212121212122e-06, "num_tokens": 1418478.0, "completions/mean_length": 170.625, "completions/min_length": 162.0, "completions/max_length": 183.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 170.625, "completions/min_terminated_length": 162.0, "completions/max_terminated_length": 183.0, "rewards/meter/mean": 0.9953961968421936, "rewards/meter/std": 0.0023201238363981247, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.2678571343421936, "rewards/repeat_penalty/std": 0.17806050181388855, "rewards/total_composite/mean": 0.2665513753890991, "rewards/total_composite/std": 0.17725923657417297, "reward": 0.2665513753890991, "reward_std": 0.17725922167301178, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01804671622812748, "sampling/sampling_logp_difference/max": 1.4201292991638184, "sampling/importance_sampling_ratio/min": 0.24168278276920319, "sampling/importance_sampling_ratio/mean": 1.0009305477142334, "sampling/importance_sampling_ratio/max": 1.6224781274795532, "entropy": 0.08471588138490915, "clip_ratio/low_mean": 0.003801907878369093, "clip_ratio/low_min": 0.003801907878369093, "clip_ratio/high_mean": 0.008620842476375401, "clip_ratio/high_max": 0.008620842476375401, "clip_ratio/region_mean": 0.012422750354744494, "reward_total_mean": 0.2665513753890991, "reward_meter_mean": 0.9953961968421936, "reward_meter_std": 0.0023201238363981247, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.2678571343421936, "reward_repeat_penalty_std": 0.17806050181388855, "reward_total_composite_mean": 0.2665513753890991, "reward_total_composite_std": 0.17725923657417297} {"timestamp_utc": "2026-04-11T22:37:30Z", "mode": "train", "global_step": 655, "epoch": 0.026308390569144878, "loss": -0.0945, "grad_norm": 2.2248940467834473, "learning_rate": 8.018181818181818e-06, "num_tokens": 1420163.0, "completions/mean_length": 116.625, "completions/min_length": 58.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 60.142860412597656, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.7503823637962341, "rewards/meter/std": 0.44998785853385925, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.5, "rewards/repeat_penalty/std": 0.30860671401023865, "rewards/total_composite/mean": 0.25364238023757935, "rewards/total_composite/std": 0.14388985931873322, "reward": 0.25364238023757935, "reward_std": 0.1438898742198944, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024109482765197754, "sampling/sampling_logp_difference/max": 1.0410680770874023, "sampling/importance_sampling_ratio/min": 0.35307735204696655, "sampling/importance_sampling_ratio/mean": 1.0044450759887695, "sampling/importance_sampling_ratio/max": 1.6566717624664307, "entropy": 0.17866009753197432, "clip_ratio/low_mean": 0.0018382353009656072, "clip_ratio/low_min": 0.0018382353009656072, "clip_ratio/high_mean": 0.012714299838989973, "clip_ratio/high_max": 0.012714299838989973, "clip_ratio/region_mean": 0.01455253513995558, "reward_total_mean": 0.25364238023757935, "reward_meter_mean": 0.7503823637962341, "reward_meter_std": 0.44998785853385925, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.5, "reward_repeat_penalty_std": 0.30860671401023865, "reward_total_composite_mean": 0.25364238023757935, "reward_total_composite_std": 0.14388985931873322} {"timestamp_utc": "2026-04-11T22:37:35Z", "mode": "train", "global_step": 656, "epoch": 0.026348556050929832, "loss": 0.0094, "grad_norm": 6.352079391479492, "learning_rate": 8.015151515151515e-06, "num_tokens": 1421891.0, "completions/mean_length": 58.0, "completions/min_length": 55.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9239930510520935, "rewards/meter/std": 0.15441425144672394, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.3333333432674408, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.3079977035522461, "rewards/total_composite/std": 0.051471415907144547, "reward": 0.3079977035522461, "reward_std": 0.05147142335772514, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022419268265366554, "sampling/sampling_logp_difference/max": 0.8770105838775635, "sampling/importance_sampling_ratio/min": 0.41602474451065063, "sampling/importance_sampling_ratio/mean": 1.0036613941192627, "sampling/importance_sampling_ratio/max": 1.928532361984253, "entropy": 0.16807558294385672, "clip_ratio/low_mean": 0.004310344811528921, "clip_ratio/low_min": 0.004310344811528921, "clip_ratio/high_mean": 0.01278492109850049, "clip_ratio/high_max": 0.01278492109850049, "clip_ratio/region_mean": 0.01709526591002941, "reward_total_mean": 0.3079977035522461, "reward_meter_mean": 0.9239930510520935, "reward_meter_std": 0.15441425144672394, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.3333333432674408, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.3079977035522461, "reward_total_composite_std": 0.051471415907144547} {"timestamp_utc": "2026-04-11T22:37:41Z", "mode": "train", "global_step": 657, "epoch": 0.026388721532714786, "loss": 0.0154, "grad_norm": 2.9878616333007812, "learning_rate": 8.012121212121214e-06, "num_tokens": 1424909.0, "completions/mean_length": 181.25, "completions/min_length": 175.0, "completions/max_length": 185.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 181.25, "completions/min_terminated_length": 175.0, "completions/max_terminated_length": 185.0, "rewards/meter/mean": 0.9874005317687988, "rewards/meter/std": 0.011436098255217075, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.20192307233810425, "rewards/repeat_penalty/std": 0.2094048410654068, "rewards/total_composite/mean": 0.1662987768650055, "rewards/total_composite/std": 0.17312301695346832, "reward": 0.1662987768650055, "reward_std": 0.17312301695346832, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010927603580057621, "sampling/sampling_logp_difference/max": 1.7186641693115234, "sampling/importance_sampling_ratio/min": 0.1793055236339569, "sampling/importance_sampling_ratio/mean": 1.0016026496887207, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.042974324664101005, "clip_ratio/low_mean": 0.008925562433432788, "clip_ratio/low_min": 0.008925562433432788, "clip_ratio/high_mean": 0.0028169237775728106, "clip_ratio/high_max": 0.0028169237775728106, "clip_ratio/region_mean": 0.011742486211005598, "reward_total_mean": 0.1662987768650055, "reward_meter_mean": 0.9874005317687988, "reward_meter_std": 0.011436098255217075, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.20192307233810425, "reward_repeat_penalty_std": 0.2094048410654068, "reward_total_composite_mean": 0.1662987768650055, "reward_total_composite_std": 0.17312301695346832} {"timestamp_utc": "2026-04-11T22:37:51Z", "mode": "train", "global_step": 658, "epoch": 0.02642888701449974, "loss": -0.0762, "grad_norm": 0.7183297276496887, "learning_rate": 8.00909090909091e-06, "num_tokens": 1429785.0, "completions/mean_length": 428.5, "completions/min_length": 394.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 416.5714416503906, "completions/min_terminated_length": 394.0, "completions/max_terminated_length": 441.0, "rewards/meter/mean": 0.8735865354537964, "rewards/meter/std": 0.35298284888267517, "rewards/count_adherence/mean": 0.6416666507720947, "rewards/count_adherence/std": 0.26170989871025085, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.17413419485092163, "rewards/repeat_penalty/std": 0.34882497787475586, "rewards/total_composite/mean": 0.033445145934820175, "rewards/total_composite/std": 0.06880706548690796, "reward": 0.033445145934820175, "reward_std": 0.06880706548690796, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0026275559794157743, "sampling/sampling_logp_difference/max": 1.0090997219085693, "sampling/importance_sampling_ratio/min": 0.3645470440387726, "sampling/importance_sampling_ratio/mean": 1.0009920597076416, "sampling/importance_sampling_ratio/max": 1.7735869884490967, "entropy": 0.012874894309788942, "clip_ratio/low_mean": 0.00030637255986221135, "clip_ratio/low_min": 0.00030637255986221135, "clip_ratio/high_mean": 0.0005817760829813778, "clip_ratio/high_max": 0.0005817760829813778, "clip_ratio/region_mean": 0.0008881486428435892, "reward_total_mean": 0.033445145934820175, "reward_meter_mean": 0.8735865354537964, "reward_meter_std": 0.35298284888267517, "reward_count_adherence_mean": 0.6416666507720947, "reward_count_adherence_std": 0.26170989871025085, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.17413419485092163, "reward_repeat_penalty_std": 0.34882497787475586, "reward_total_composite_mean": 0.033445145934820175, "reward_total_composite_std": 0.06880706548690796} {"timestamp_utc": "2026-04-11T22:38:00Z", "mode": "train", "global_step": 659, "epoch": 0.026469052496284694, "loss": -0.1652, "grad_norm": 1.3609012365341187, "learning_rate": 8.006060606060607e-06, "num_tokens": 1431540.0, "completions/mean_length": 190.375, "completions/min_length": 81.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 83.16667175292969, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.9505882263183594, "rewards/meter/std": 0.030698692426085472, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 0.7916666865348816, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.513268232345581, "rewards/total_composite/std": 0.3378344178199768, "reward": 0.513268232345581, "reward_std": 0.3378343880176544, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02022649347782135, "sampling/sampling_logp_difference/max": 1.8950445652008057, "sampling/importance_sampling_ratio/min": 0.1503116339445114, "sampling/importance_sampling_ratio/mean": 1.0079931020736694, "sampling/importance_sampling_ratio/max": 1.7507625818252563, "entropy": 0.06744233565405011, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.008969587041065097, "clip_ratio/high_max": 0.008969587041065097, "clip_ratio/region_mean": 0.008969587041065097, "reward_total_mean": 0.513268232345581, "reward_meter_mean": 0.9505882263183594, "reward_meter_std": 0.030698692426085472, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 0.7916666865348816, "reward_repeat_penalty_std": 0.17251639068126678, "reward_total_composite_mean": 0.513268232345581, "reward_total_composite_std": 0.3378344178199768} {"timestamp_utc": "2026-04-11T22:38:05Z", "mode": "train", "global_step": 660, "epoch": 0.026509217978069648, "loss": 0.0308, "grad_norm": 2.2579872608184814, "learning_rate": 8.003030303030304e-06, "num_tokens": 1433681.0, "completions/mean_length": 110.625, "completions/min_length": 104.0, "completions/max_length": 138.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 110.625, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.9927444458007812, "rewards/meter/std": 0.0012347318697720766, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.23571428656578064, "rewards/repeat_penalty/std": 0.07284314185380936, "rewards/total_composite/mean": 0.22219520807266235, "rewards/total_composite/std": 0.07082200050354004, "reward": 0.22219520807266235, "reward_std": 0.07082199305295944, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008687403053045273, "sampling/sampling_logp_difference/max": 1.8446245193481445, "sampling/importance_sampling_ratio/min": 0.1580846756696701, "sampling/importance_sampling_ratio/mean": 1.001028299331665, "sampling/importance_sampling_ratio/max": 1.6747108697891235, "entropy": 0.020129066659137607, "clip_ratio/low_mean": 0.006200430449098349, "clip_ratio/low_min": 0.006200430449098349, "clip_ratio/high_mean": 0.0036057692486792803, "clip_ratio/high_max": 0.0036057692486792803, "clip_ratio/region_mean": 0.009806199697777629, "reward_total_mean": 0.22219520807266235, "reward_meter_mean": 0.9927444458007812, "reward_meter_std": 0.0012347318697720766, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.23571428656578064, "reward_repeat_penalty_std": 0.07284314185380936, "reward_total_composite_mean": 0.22219520807266235, "reward_total_composite_std": 0.07082200050354004} {"timestamp_utc": "2026-04-11T22:38:15Z", "mode": "train", "global_step": 661, "epoch": 0.0265493834598546, "loss": -0.0515, "grad_norm": 3.181408405303955, "learning_rate": 8.000000000000001e-06, "num_tokens": 1435750.0, "completions/mean_length": 203.625, "completions/min_length": 81.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 100.83333587646484, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.43306416273117065, "rewards/meter/std": 0.44791871309280396, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.18600596487522125, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 0.6357142925262451, "rewards/repeat_penalty/std": 0.31916436553001404, "rewards/total_composite/mean": 0.22007080912590027, "rewards/total_composite/std": 0.2968771755695343, "reward": 0.22007080912590027, "reward_std": 0.2968771457672119, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04375706985592842, "sampling/sampling_logp_difference/max": 1.372495174407959, "sampling/importance_sampling_ratio/min": 0.253473699092865, "sampling/importance_sampling_ratio/mean": 1.0068386793136597, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17050567921251059, "clip_ratio/low_mean": 0.008919409476220608, "clip_ratio/low_min": 0.008919409476220608, "clip_ratio/high_mean": 0.017718179151415825, "clip_ratio/high_max": 0.017718179151415825, "clip_ratio/region_mean": 0.026637588627636433, "reward_total_mean": 0.22007080912590027, "reward_meter_mean": 0.43306416273117065, "reward_meter_std": 0.44791871309280396, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.18600596487522125, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 0.6357142925262451, "reward_repeat_penalty_std": 0.31916436553001404, "reward_total_composite_mean": 0.22007080912590027, "reward_total_composite_std": 0.2968771755695343} {"timestamp_utc": "2026-04-11T22:38:25Z", "mode": "train", "global_step": 662, "epoch": 0.026589548941639556, "loss": -0.1279, "grad_norm": 2.6338069438934326, "learning_rate": 7.996969696969697e-06, "num_tokens": 1437638.0, "completions/mean_length": 117.0, "completions/min_length": 57.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 60.57143020629883, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9266777634620667, "rewards/meter/std": 0.1883465051651001, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.5833333730697632, "rewards/repeat_penalty/std": 0.2357022762298584, "rewards/total_composite/mean": 0.455108642578125, "rewards/total_composite/std": 0.24599777162075043, "reward": 0.455108642578125, "reward_std": 0.24599777162075043, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.032812319695949554, "sampling/sampling_logp_difference/max": 1.3324346542358398, "sampling/importance_sampling_ratio/min": 0.263834148645401, "sampling/importance_sampling_ratio/mean": 1.0021228790283203, "sampling/importance_sampling_ratio/max": 1.474334478378296, "entropy": 0.17267457023262978, "clip_ratio/low_mean": 0.004385964944958687, "clip_ratio/low_min": 0.004385964944958687, "clip_ratio/high_mean": 0.012073024990968406, "clip_ratio/high_max": 0.012073024990968406, "clip_ratio/region_mean": 0.016458989935927093, "reward_total_mean": 0.455108642578125, "reward_meter_mean": 0.9266777634620667, "reward_meter_std": 0.1883465051651001, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.5833333730697632, "reward_repeat_penalty_std": 0.2357022762298584, "reward_total_composite_mean": 0.455108642578125, "reward_total_composite_std": 0.24599777162075043} {"timestamp_utc": "2026-04-11T22:38:30Z", "mode": "train", "global_step": 663, "epoch": 0.02662971442342451, "loss": 0.021, "grad_norm": 4.197593688964844, "learning_rate": 7.993939393939396e-06, "num_tokens": 1439950.0, "completions/mean_length": 108.0, "completions/min_length": 96.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.0, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.9684537053108215, "rewards/meter/std": 0.027583837509155273, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4226190447807312, "rewards/repeat_penalty/std": 0.2210753709077835, "rewards/total_composite/mean": 0.3900730013847351, "rewards/total_composite/std": 0.19891957938671112, "reward": 0.3900730013847351, "reward_std": 0.19891956448554993, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02085985243320465, "sampling/sampling_logp_difference/max": 1.0295929908752441, "sampling/importance_sampling_ratio/min": 0.3571523129940033, "sampling/importance_sampling_ratio/mean": 0.9997091889381409, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11175651382654905, "clip_ratio/low_mean": 0.00461137923412025, "clip_ratio/low_min": 0.00461137923412025, "clip_ratio/high_mean": 0.014703674940392375, "clip_ratio/high_max": 0.014703674940392375, "clip_ratio/region_mean": 0.019315054174512625, "reward_total_mean": 0.3900730013847351, "reward_meter_mean": 0.9684537053108215, "reward_meter_std": 0.027583837509155273, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4226190447807312, "reward_repeat_penalty_std": 0.2210753709077835, "reward_total_composite_mean": 0.3900730013847351, "reward_total_composite_std": 0.19891957938671112} {"timestamp_utc": "2026-04-11T22:38:35Z", "mode": "train", "global_step": 664, "epoch": 0.026669879905209463, "loss": -0.029, "grad_norm": 7.99793004989624, "learning_rate": 7.990909090909091e-06, "num_tokens": 1441763.0, "completions/mean_length": 70.625, "completions/min_length": 68.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.625, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9828510880470276, "rewards/meter/std": 0.025825461372733116, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5833333730697632, "rewards/repeat_penalty/std": 0.29546841979026794, "rewards/total_composite/mean": 0.568142294883728, "rewards/total_composite/std": 0.27455055713653564, "reward": 0.568142294883728, "reward_std": 0.27455058693885803, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03234035521745682, "sampling/sampling_logp_difference/max": 1.1230463981628418, "sampling/importance_sampling_ratio/min": 0.3252873122692108, "sampling/importance_sampling_ratio/mean": 1.0028775930404663, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10323597816750407, "clip_ratio/low_mean": 0.0055147059028968215, "clip_ratio/low_min": 0.0055147059028968215, "clip_ratio/high_mean": 0.023867564275860786, "clip_ratio/high_max": 0.023867564275860786, "clip_ratio/region_mean": 0.029382270178757608, "reward_total_mean": 0.568142294883728, "reward_meter_mean": 0.9828510880470276, "reward_meter_std": 0.025825461372733116, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5833333730697632, "reward_repeat_penalty_std": 0.29546841979026794, "reward_total_composite_mean": 0.568142294883728, "reward_total_composite_std": 0.27455055713653564} {"timestamp_utc": "2026-04-11T22:38:39Z", "mode": "train", "global_step": 665, "epoch": 0.026710045386994417, "loss": -0.0097, "grad_norm": 3.2300848960876465, "learning_rate": 7.987878787878789e-06, "num_tokens": 1443616.0, "completions/mean_length": 80.625, "completions/min_length": 74.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.625, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.9603564143180847, "rewards/meter/std": 0.023673059418797493, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7916666865348816, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.7601395845413208, "rewards/total_composite/std": 0.16597026586532593, "reward": 0.7601395845413208, "reward_std": 0.16597026586532593, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0124077582731843, "sampling/sampling_logp_difference/max": 1.0308361053466797, "sampling/importance_sampling_ratio/min": 0.3567086160182953, "sampling/importance_sampling_ratio/mean": 1.0014386177062988, "sampling/importance_sampling_ratio/max": 1.564980387687683, "entropy": 0.07347342604771256, "clip_ratio/low_mean": 0.009405238670296967, "clip_ratio/low_min": 0.009405238670296967, "clip_ratio/high_mean": 0.0015432098880410194, "clip_ratio/high_max": 0.0015432098880410194, "clip_ratio/region_mean": 0.010948448558337986, "reward_total_mean": 0.7601395845413208, "reward_meter_mean": 0.9603564143180847, "reward_meter_std": 0.023673059418797493, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7916666865348816, "reward_repeat_penalty_std": 0.17251639068126678, "reward_total_composite_mean": 0.7601395845413208, "reward_total_composite_std": 0.16597026586532593} {"timestamp_utc": "2026-04-11T22:38:44Z", "mode": "train", "global_step": 666, "epoch": 0.02675021086877937, "loss": 0.0007, "grad_norm": 3.160210609436035, "learning_rate": 7.984848484848486e-06, "num_tokens": 1445930.0, "completions/mean_length": 122.25, "completions/min_length": 120.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.25, "completions/min_terminated_length": 120.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.8794082999229431, "rewards/meter/std": 0.32782238721847534, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.212053582072258, "rewards/repeat_penalty/std": 0.2030286192893982, "rewards/total_composite/mean": 0.12827160954475403, "rewards/total_composite/std": 0.03277355059981346, "reward": 0.12827160954475403, "reward_std": 0.03277355059981346, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02072002924978733, "sampling/sampling_logp_difference/max": 6.419788360595703, "sampling/importance_sampling_ratio/min": 0.0016290009953081608, "sampling/importance_sampling_ratio/mean": 0.9980396032333374, "sampling/importance_sampling_ratio/max": 1.6665557622909546, "entropy": 0.03877314692363143, "clip_ratio/low_mean": 0.0010416667209938169, "clip_ratio/low_min": 0.0010416667209938169, "clip_ratio/high_mean": 0.006232782383449376, "clip_ratio/high_max": 0.006232782383449376, "clip_ratio/region_mean": 0.0072744491044431925, "reward_total_mean": 0.12827160954475403, "reward_meter_mean": 0.8794082999229431, "reward_meter_std": 0.32782238721847534, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.212053582072258, "reward_repeat_penalty_std": 0.2030286192893982, "reward_total_composite_mean": 0.12827160954475403, "reward_total_composite_std": 0.03277355059981346} {"timestamp_utc": "2026-04-11T22:38:50Z", "mode": "train", "global_step": 667, "epoch": 0.026790376350564325, "loss": 0.0108, "grad_norm": 4.775012016296387, "learning_rate": 7.981818181818183e-06, "num_tokens": 1448313.0, "completions/mean_length": 120.875, "completions/min_length": 116.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.875, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9965612888336182, "rewards/meter/std": 0.0018690497381612659, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.25, "rewards/repeat_penalty/std": 0.21257823705673218, "rewards/total_composite/mean": 0.24911293387413025, "rewards/total_composite/std": 0.2116239219903946, "reward": 0.24911293387413025, "reward_std": 0.2116239219903946, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017011119052767754, "sampling/sampling_logp_difference/max": 1.0500693321228027, "sampling/importance_sampling_ratio/min": 0.3773258924484253, "sampling/importance_sampling_ratio/mean": 0.9989076256752014, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07529849279671907, "clip_ratio/low_mean": 0.0031968391267582774, "clip_ratio/low_min": 0.0031968391267582774, "clip_ratio/high_mean": 0.006372183095663786, "clip_ratio/high_max": 0.006372183095663786, "clip_ratio/region_mean": 0.009569022222422063, "reward_total_mean": 0.24911293387413025, "reward_meter_mean": 0.9965612888336182, "reward_meter_std": 0.0018690497381612659, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.25, "reward_repeat_penalty_std": 0.21257823705673218, "reward_total_composite_mean": 0.24911293387413025, "reward_total_composite_std": 0.2116239219903946} {"timestamp_utc": "2026-04-11T22:38:59Z", "mode": "train", "global_step": 668, "epoch": 0.02683054183234928, "loss": -0.1559, "grad_norm": 2.0396084785461426, "learning_rate": 7.978787878787879e-06, "num_tokens": 1451329.0, "completions/mean_length": 243.0, "completions/min_length": 195.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 204.57144165039062, "completions/min_terminated_length": 195.0, "completions/max_terminated_length": 207.0, "rewards/meter/mean": 0.9733182787895203, "rewards/meter/std": 0.02887692302465439, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.27272728085517883, "rewards/repeat_penalty/std": 0.13744163513183594, "rewards/total_composite/mean": 0.17570531368255615, "rewards/total_composite/std": 0.11893084645271301, "reward": 0.17570531368255615, "reward_std": 0.11893083900213242, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01857808604836464, "sampling/sampling_logp_difference/max": 2.0781538486480713, "sampling/importance_sampling_ratio/min": 0.1251610666513443, "sampling/importance_sampling_ratio/mean": 0.9996070861816406, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.05400602100417018, "clip_ratio/low_mean": 0.007991038146428764, "clip_ratio/low_min": 0.007991038146428764, "clip_ratio/high_mean": 0.0030251864809542894, "clip_ratio/high_max": 0.0030251864809542894, "clip_ratio/region_mean": 0.011016224627383053, "reward_total_mean": 0.17570531368255615, "reward_meter_mean": 0.9733182787895203, "reward_meter_std": 0.02887692302465439, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.27272728085517883, "reward_repeat_penalty_std": 0.13744163513183594, "reward_total_composite_mean": 0.17570531368255615, "reward_total_composite_std": 0.11893084645271301} {"timestamp_utc": "2026-04-11T22:39:09Z", "mode": "train", "global_step": 669, "epoch": 0.026870707314134233, "loss": -0.023, "grad_norm": 2.1599395275115967, "learning_rate": 7.975757575757576e-06, "num_tokens": 1453130.0, "completions/mean_length": 240.125, "completions/min_length": 74.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 77.0, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.5011062026023865, "rewards/meter/std": 0.4051038920879364, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4261804223060608, "rewards/total_composite/std": 0.4667132496833801, "reward": 0.4261804223060608, "reward_std": 0.46671321988105774, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.052589599043130875, "sampling/sampling_logp_difference/max": 1.824247121810913, "sampling/importance_sampling_ratio/min": 0.16133907437324524, "sampling/importance_sampling_ratio/mean": 1.0080572366714478, "sampling/importance_sampling_ratio/max": 1.8760875463485718, "entropy": 0.16993126086890697, "clip_ratio/low_mean": 0.009377967799082398, "clip_ratio/low_min": 0.009377967799082398, "clip_ratio/high_mean": 0.011646514758467674, "clip_ratio/high_max": 0.011646514758467674, "clip_ratio/region_mean": 0.021024482557550073, "reward_total_mean": 0.4261804223060608, "reward_meter_mean": 0.5011062026023865, "reward_meter_std": 0.4051038920879364, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.4261804223060608, "reward_total_composite_std": 0.4667132496833801} {"timestamp_utc": "2026-04-11T22:39:18Z", "mode": "train", "global_step": 670, "epoch": 0.026910872795919187, "loss": -0.1072, "grad_norm": 0.766589879989624, "learning_rate": 7.972727272727273e-06, "num_tokens": 1454698.0, "completions/mean_length": 351.0, "completions/min_length": 73.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.625, "completions/mean_terminated_length": 82.66667175292969, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.6174333691596985, "rewards/meter/std": 0.4055008888244629, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.375, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/repeat_penalty/mean": 0.7083333730697632, "rewards/repeat_penalty/std": 0.11785111576318741, "rewards/total_composite/mean": 0.249087393283844, "rewards/total_composite/std": 0.3437734544277191, "reward": 0.249087393283844, "reward_std": 0.3437734842300415, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03491988405585289, "sampling/sampling_logp_difference/max": 1.1255141496658325, "sampling/importance_sampling_ratio/min": 0.4397118091583252, "sampling/importance_sampling_ratio/mean": 1.004920482635498, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09451978467404842, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.013186233583837748, "clip_ratio/high_max": 0.013186233583837748, "clip_ratio/region_mean": 0.013186233583837748, "reward_total_mean": 0.249087393283844, "reward_meter_mean": 0.6174333691596985, "reward_meter_std": 0.4055008888244629, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.375, "reward_arabic_clean_std": 0.5175492167472839, "reward_repeat_penalty_mean": 0.7083333730697632, "reward_repeat_penalty_std": 0.11785111576318741, "reward_total_composite_mean": 0.249087393283844, "reward_total_composite_std": 0.3437734544277191} {"timestamp_utc": "2026-04-11T22:39:28Z", "mode": "train", "global_step": 671, "epoch": 0.02695103827770414, "loss": -0.0912, "grad_norm": 0.7819209098815918, "learning_rate": 7.96969696969697e-06, "num_tokens": 1456328.0, "completions/mean_length": 354.75, "completions/min_length": 91.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.625, "completions/mean_terminated_length": 92.66667175292969, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.7872094511985779, "rewards/meter/std": 0.3365079164505005, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 0.5, "rewards/arabic_clean/std": 0.5345224738121033, "rewards/repeat_penalty/mean": 0.4750000238418579, "rewards/repeat_penalty/std": 0.2121320366859436, "rewards/total_composite/mean": 0.19879300892353058, "rewards/total_composite/std": 0.21251867711544037, "reward": 0.19879300892353058, "reward_std": 0.21251867711544037, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0315895713865757, "sampling/sampling_logp_difference/max": 1.163419246673584, "sampling/importance_sampling_ratio/min": 0.31241610646247864, "sampling/importance_sampling_ratio/mean": 1.0077892541885376, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07205967605113983, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.006735671544447541, "clip_ratio/high_max": 0.006735671544447541, "clip_ratio/region_mean": 0.006735671544447541, "reward_total_mean": 0.19879300892353058, "reward_meter_mean": 0.7872094511985779, "reward_meter_std": 0.3365079164505005, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 0.5, "reward_arabic_clean_std": 0.5345224738121033, "reward_repeat_penalty_mean": 0.4750000238418579, "reward_repeat_penalty_std": 0.2121320366859436, "reward_total_composite_mean": 0.19879300892353058, "reward_total_composite_std": 0.21251867711544037} {"timestamp_utc": "2026-04-11T22:39:38Z", "mode": "train", "global_step": 672, "epoch": 0.026991203759489095, "loss": -0.1237, "grad_norm": 0.8036666512489319, "learning_rate": 7.966666666666668e-06, "num_tokens": 1458728.0, "completions/mean_length": 314.0, "completions/min_length": 160.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 195.1999969482422, "completions/min_terminated_length": 160.0, "completions/max_terminated_length": 220.0, "rewards/meter/mean": 0.7117201685905457, "rewards/meter/std": 0.3770325481891632, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.1414213627576828, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 0.5681818127632141, "rewards/repeat_penalty/std": 0.2368127554655075, "rewards/total_composite/mean": 0.23602712154388428, "rewards/total_composite/std": 0.1749935895204544, "reward": 0.23602712154388428, "reward_std": 0.1749935895204544, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020951304584741592, "sampling/sampling_logp_difference/max": 1.4866762161254883, "sampling/importance_sampling_ratio/min": 0.22612299025058746, "sampling/importance_sampling_ratio/mean": 0.9994598627090454, "sampling/importance_sampling_ratio/max": 1.857919692993164, "entropy": 0.06148350611329079, "clip_ratio/low_mean": 0.0028696630615741014, "clip_ratio/low_min": 0.0028696630615741014, "clip_ratio/high_mean": 0.010866477387025952, "clip_ratio/high_max": 0.010866477387025952, "clip_ratio/region_mean": 0.013736140448600054, "reward_total_mean": 0.23602712154388428, "reward_meter_mean": 0.7117201685905457, "reward_meter_std": 0.3770325481891632, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.1414213627576828, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 0.5681818127632141, "reward_repeat_penalty_std": 0.2368127554655075, "reward_total_composite_mean": 0.23602712154388428, "reward_total_composite_std": 0.1749935895204544} {"timestamp_utc": "2026-04-11T22:39:48Z", "mode": "train", "global_step": 673, "epoch": 0.02703136924127405, "loss": -0.0271, "grad_norm": 4.668978214263916, "learning_rate": 7.963636363636365e-06, "num_tokens": 1460405.0, "completions/mean_length": 117.625, "completions/min_length": 58.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 61.28571701049805, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.8629885911941528, "rewards/meter/std": 0.28332778811454773, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.768899142742157, "rewards/total_composite/std": 0.3212806284427643, "reward": 0.768899142742157, "reward_std": 0.32128065824508667, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04510435834527016, "sampling/sampling_logp_difference/max": 1.6506894826889038, "sampling/importance_sampling_ratio/min": 0.1919175386428833, "sampling/importance_sampling_ratio/mean": 1.0077714920043945, "sampling/importance_sampling_ratio/max": 1.7451122999191284, "entropy": 0.2579981219023466, "clip_ratio/low_mean": 0.020833334419876337, "clip_ratio/low_min": 0.020833334419876337, "clip_ratio/high_mean": 0.02129602595232427, "clip_ratio/high_max": 0.02129602595232427, "clip_ratio/region_mean": 0.04212936037220061, "reward_total_mean": 0.768899142742157, "reward_meter_mean": 0.8629885911941528, "reward_meter_std": 0.28332778811454773, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.768899142742157, "reward_total_composite_std": 0.3212806284427643} {"timestamp_utc": "2026-04-11T22:39:58Z", "mode": "train", "global_step": 674, "epoch": 0.027071534723059003, "loss": -0.0956, "grad_norm": 3.691714286804199, "learning_rate": 7.96060606060606e-06, "num_tokens": 1462244.0, "completions/mean_length": 128.875, "completions/min_length": 69.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 74.14286041259766, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.7070336937904358, "rewards/meter/std": 0.39935943484306335, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.8333333730697632, "rewards/repeat_penalty/std": 0.17817415297031403, "rewards/total_composite/mean": 0.5691444873809814, "rewards/total_composite/std": 0.343730628490448, "reward": 0.5691444873809814, "reward_std": 0.3437305986881256, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05939958617091179, "sampling/sampling_logp_difference/max": 1.2720730304718018, "sampling/importance_sampling_ratio/min": 0.28025004267692566, "sampling/importance_sampling_ratio/mean": 1.0029557943344116, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23062355443835258, "clip_ratio/low_mean": 0.009934040834195912, "clip_ratio/low_min": 0.009934040834195912, "clip_ratio/high_mean": 0.04135313397273421, "clip_ratio/high_max": 0.04135313397273421, "clip_ratio/region_mean": 0.051287174806930125, "reward_total_mean": 0.5691444873809814, "reward_meter_mean": 0.7070336937904358, "reward_meter_std": 0.39935943484306335, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.8333333730697632, "reward_repeat_penalty_std": 0.17817415297031403, "reward_total_composite_mean": 0.5691444873809814, "reward_total_composite_std": 0.343730628490448} {"timestamp_utc": "2026-04-11T22:40:08Z", "mode": "train", "global_step": 675, "epoch": 0.027111700204843957, "loss": -0.0973, "grad_norm": 0.7685384154319763, "learning_rate": 7.957575757575758e-06, "num_tokens": 1466421.0, "completions/mean_length": 376.125, "completions/min_length": 337.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 356.71429443359375, "completions/min_terminated_length": 337.0, "completions/max_terminated_length": 367.0, "rewards/meter/mean": 0.9768602848052979, "rewards/meter/std": 0.0202656090259552, "rewards/count_adherence/mean": 0.8624999523162842, "rewards/count_adherence/std": 0.31139087677001953, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.2991071343421936, "rewards/repeat_penalty/std": 0.36836329102516174, "rewards/total_composite/mean": 0.15836204588413239, "rewards/total_composite/std": 0.21933622658252716, "reward": 0.15836204588413239, "reward_std": 0.21933621168136597, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005116640590131283, "sampling/sampling_logp_difference/max": 1.467843770980835, "sampling/importance_sampling_ratio/min": 0.4014633893966675, "sampling/importance_sampling_ratio/mean": 1.0009857416152954, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0205289286095649, "clip_ratio/low_mean": 0.0032165506563615054, "clip_ratio/low_min": 0.0032165506563615054, "clip_ratio/high_mean": 0.000681198900565505, "clip_ratio/high_max": 0.000681198900565505, "clip_ratio/region_mean": 0.0038977495569270104, "reward_total_mean": 0.15836204588413239, "reward_meter_mean": 0.9768602848052979, "reward_meter_std": 0.0202656090259552, "reward_count_adherence_mean": 0.8624999523162842, "reward_count_adherence_std": 0.31139087677001953, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.2991071343421936, "reward_repeat_penalty_std": 0.36836329102516174, "reward_total_composite_mean": 0.15836204588413239, "reward_total_composite_std": 0.21933622658252716} {"timestamp_utc": "2026-04-11T22:40:12Z", "mode": "train", "global_step": 676, "epoch": 0.02715186568662891, "loss": -0.1218, "grad_norm": 9.489598274230957, "learning_rate": 7.954545454545455e-06, "num_tokens": 1468025.0, "completions/mean_length": 45.5, "completions/min_length": 41.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.5, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9916675090789795, "rewards/meter/std": 0.0027789692394435406, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9916675090789795, "rewards/total_composite/std": 0.0027789692394435406, "reward": 0.9916675090789795, "reward_std": 0.002778968308120966, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.040273357182741165, "sampling/sampling_logp_difference/max": 2.040062665939331, "sampling/importance_sampling_ratio/min": 0.13002057373523712, "sampling/importance_sampling_ratio/mean": 0.9964287877082825, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.16463111247867346, "clip_ratio/low_mean": 0.009073751280084252, "clip_ratio/low_min": 0.009073751280084252, "clip_ratio/high_mean": 0.01704174862243235, "clip_ratio/high_max": 0.01704174862243235, "clip_ratio/region_mean": 0.026115499902516603, "reward_total_mean": 0.9916675090789795, "reward_meter_mean": 0.9916675090789795, "reward_meter_std": 0.0027789692394435406, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9916675090789795, "reward_total_composite_std": 0.0027789692394435406} {"timestamp_utc": "2026-04-11T22:40:17Z", "mode": "train", "global_step": 677, "epoch": 0.027192031168413865, "loss": 0.0185, "grad_norm": 4.125744342803955, "learning_rate": 7.951515151515152e-06, "num_tokens": 1469690.0, "completions/mean_length": 62.125, "completions/min_length": 58.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.125, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9938265085220337, "rewards/meter/std": 0.0003937912406399846, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6625509858131409, "rewards/total_composite/std": 0.00026252749375998974, "reward": 0.6625509858131409, "reward_std": 0.0002625406195875257, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02882186695933342, "sampling/sampling_logp_difference/max": 5.798013687133789, "sampling/importance_sampling_ratio/min": 0.0030335744377225637, "sampling/importance_sampling_ratio/mean": 1.0004545450210571, "sampling/importance_sampling_ratio/max": 1.5561105012893677, "entropy": 0.11069364938884974, "clip_ratio/low_mean": 0.012098872568458319, "clip_ratio/low_min": 0.012098872568458319, "clip_ratio/high_mean": 0.0021551724057644606, "clip_ratio/high_max": 0.0021551724057644606, "clip_ratio/region_mean": 0.01425404497422278, "reward_total_mean": 0.6625509858131409, "reward_meter_mean": 0.9938265085220337, "reward_meter_std": 0.0003937912406399846, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6625509858131409, "reward_total_composite_std": 0.00026252749375998974} {"timestamp_utc": "2026-04-11T22:40:27Z", "mode": "train", "global_step": 678, "epoch": 0.02723219665019882, "loss": -0.15, "grad_norm": 0.7712501287460327, "learning_rate": 7.948484848484848e-06, "num_tokens": 1472788.0, "completions/mean_length": 336.25, "completions/min_length": 268.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 277.66668701171875, "completions/min_terminated_length": 268.0, "completions/max_terminated_length": 291.0, "rewards/meter/mean": 0.6280694007873535, "rewards/meter/std": 0.5047115683555603, "rewards/count_adherence/mean": 0.8035714626312256, "rewards/count_adherence/std": 0.15152288973331451, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 0.41633522510528564, "rewards/repeat_penalty/std": 0.3482770323753357, "rewards/total_composite/mean": 0.1397983729839325, "rewards/total_composite/std": 0.1796947717666626, "reward": 0.1397983729839325, "reward_std": 0.1796947568655014, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01377029623836279, "sampling/sampling_logp_difference/max": 5.731827259063721, "sampling/importance_sampling_ratio/min": 0.003241149475798011, "sampling/importance_sampling_ratio/mean": 0.9978663921356201, "sampling/importance_sampling_ratio/max": 1.6066981554031372, "entropy": 0.03139376197941601, "clip_ratio/low_mean": 0.004044867469929159, "clip_ratio/low_min": 0.004044867469929159, "clip_ratio/high_mean": 0.002251501166028902, "clip_ratio/high_max": 0.002251501166028902, "clip_ratio/region_mean": 0.006296368635958061, "reward_total_mean": 0.1397983729839325, "reward_meter_mean": 0.6280694007873535, "reward_meter_std": 0.5047115683555603, "reward_count_adherence_mean": 0.8035714626312256, "reward_count_adherence_std": 0.15152288973331451, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 0.41633522510528564, "reward_repeat_penalty_std": 0.3482770323753357, "reward_total_composite_mean": 0.1397983729839325, "reward_total_composite_std": 0.1796947717666626} {"timestamp_utc": "2026-04-11T22:40:31Z", "mode": "train", "global_step": 679, "epoch": 0.027272362131983773, "loss": -0.0123, "grad_norm": 12.517946243286133, "learning_rate": 7.945454545454547e-06, "num_tokens": 1474604.0, "completions/mean_length": 57.0, "completions/min_length": 53.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.0, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9841916561126709, "rewards/meter/std": 0.005828971043229103, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9428657293319702, "rewards/total_composite/std": 0.1139112040400505, "reward": 0.9428657293319702, "reward_std": 0.1139112189412117, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02025281824171543, "sampling/sampling_logp_difference/max": 1.1112747192382812, "sampling/importance_sampling_ratio/min": 0.3291391432285309, "sampling/importance_sampling_ratio/mean": 0.9998968243598938, "sampling/importance_sampling_ratio/max": 1.880238652229309, "entropy": 0.11628487333655357, "clip_ratio/low_mean": 0.002358490601181984, "clip_ratio/low_min": 0.002358490601181984, "clip_ratio/high_mean": 0.01512786210514605, "clip_ratio/high_max": 0.01512786210514605, "clip_ratio/region_mean": 0.017486352706328034, "reward_total_mean": 0.9428657293319702, "reward_meter_mean": 0.9841916561126709, "reward_meter_std": 0.005828971043229103, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.9428657293319702, "reward_total_composite_std": 0.1139112040400505} {"timestamp_utc": "2026-04-11T22:40:36Z", "mode": "train", "global_step": 680, "epoch": 0.027312527613768726, "loss": 0.031, "grad_norm": 8.33311653137207, "learning_rate": 7.942424242424242e-06, "num_tokens": 1476777.0, "completions/mean_length": 115.625, "completions/min_length": 101.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.625, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.371232271194458, "rewards/meter/std": 0.47603076696395874, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.48125001788139343, "rewards/repeat_penalty/std": 0.13611315190792084, "rewards/total_composite/mean": 0.22119268774986267, "rewards/total_composite/std": 0.28687217831611633, "reward": 0.22119268774986267, "reward_std": 0.2868722081184387, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020960384979844093, "sampling/sampling_logp_difference/max": 1.9988516569137573, "sampling/importance_sampling_ratio/min": 0.1354907900094986, "sampling/importance_sampling_ratio/mean": 1.0001816749572754, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08438334474340081, "clip_ratio/low_mean": 0.014545571291819215, "clip_ratio/low_min": 0.014545571291819215, "clip_ratio/high_mean": 0.0033385155256837606, "clip_ratio/high_max": 0.0033385155256837606, "clip_ratio/region_mean": 0.017884086817502975, "reward_total_mean": 0.22119268774986267, "reward_meter_mean": 0.371232271194458, "reward_meter_std": 0.47603076696395874, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.48125001788139343, "reward_repeat_penalty_std": 0.13611315190792084, "reward_total_composite_mean": 0.22119268774986267, "reward_total_composite_std": 0.28687217831611633} {"timestamp_utc": "2026-04-11T22:40:41Z", "mode": "train", "global_step": 681, "epoch": 0.02735269309555368, "loss": 0.0231, "grad_norm": 3.9222095012664795, "learning_rate": 7.93939393939394e-06, "num_tokens": 1478560.0, "completions/mean_length": 57.875, "completions/min_length": 57.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.875, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9723160266876221, "rewards/meter/std": 0.027899622917175293, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.8934845924377441, "rewards/total_composite/std": 0.1633896678686142, "reward": 0.8934845924377441, "reward_std": 0.163389652967453, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023344971239566803, "sampling/sampling_logp_difference/max": 1.0728743076324463, "sampling/importance_sampling_ratio/min": 0.342024028301239, "sampling/importance_sampling_ratio/mean": 1.0034892559051514, "sampling/importance_sampling_ratio/max": 1.6929512023925781, "entropy": 0.11835484858602285, "clip_ratio/low_mean": 0.014689265750348568, "clip_ratio/low_min": 0.014689265750348568, "clip_ratio/high_mean": 0.015127861872315407, "clip_ratio/high_max": 0.015127861872315407, "clip_ratio/region_mean": 0.029817127622663975, "reward_total_mean": 0.8934845924377441, "reward_meter_mean": 0.9723160266876221, "reward_meter_std": 0.027899622917175293, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_total_composite_mean": 0.8934845924377441, "reward_total_composite_std": 0.1633896678686142} {"timestamp_utc": "2026-04-11T22:40:50Z", "mode": "train", "global_step": 682, "epoch": 0.027392858577338634, "loss": -0.0676, "grad_norm": 1.2039419412612915, "learning_rate": 7.936363636363637e-06, "num_tokens": 1480663.0, "completions/mean_length": 173.875, "completions/min_length": 109.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 125.5714340209961, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.38672682642936707, "rewards/meter/std": 0.36102262139320374, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.2519763112068176, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.48750001192092896, "rewards/repeat_penalty/std": 0.24604006111621857, "rewards/total_composite/mean": 0.16035160422325134, "rewards/total_composite/std": 0.24009783565998077, "reward": 0.16035160422325134, "reward_std": 0.24009782075881958, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013726767152547836, "sampling/sampling_logp_difference/max": 1.2122516632080078, "sampling/importance_sampling_ratio/min": 0.29752659797668457, "sampling/importance_sampling_ratio/mean": 1.0000630617141724, "sampling/importance_sampling_ratio/max": 1.6329307556152344, "entropy": 0.049124513287097216, "clip_ratio/low_mean": 0.005251961061730981, "clip_ratio/low_min": 0.005251961061730981, "clip_ratio/high_mean": 0.0029069767333567142, "clip_ratio/high_max": 0.0029069767333567142, "clip_ratio/region_mean": 0.008158937795087695, "reward_total_mean": 0.16035160422325134, "reward_meter_mean": 0.38672682642936707, "reward_meter_std": 0.36102262139320374, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.2519763112068176, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.48750001192092896, "reward_repeat_penalty_std": 0.24604006111621857, "reward_total_composite_mean": 0.16035160422325134, "reward_total_composite_std": 0.24009783565998077} {"timestamp_utc": "2026-04-11T22:40:55Z", "mode": "train", "global_step": 683, "epoch": 0.02743302405912359, "loss": 0.0096, "grad_norm": 4.130276679992676, "learning_rate": 7.933333333333334e-06, "num_tokens": 1482349.0, "completions/mean_length": 58.75, "completions/min_length": 57.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.75, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9239417314529419, "rewards/meter/std": 0.05428864806890488, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.8474001884460449, "rewards/total_composite/std": 0.15459680557250977, "reward": 0.8474001884460449, "reward_std": 0.15459680557250977, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03419341892004013, "sampling/sampling_logp_difference/max": 1.8840758800506592, "sampling/importance_sampling_ratio/min": 0.15196943283081055, "sampling/importance_sampling_ratio/mean": 0.994130551815033, "sampling/importance_sampling_ratio/max": 1.6103789806365967, "entropy": 0.11382754053920507, "clip_ratio/low_mean": 0.0042372881434857845, "clip_ratio/low_min": 0.0042372881434857845, "clip_ratio/high_mean": 0.016894312808290124, "clip_ratio/high_max": 0.016894312808290124, "clip_ratio/region_mean": 0.02113160095177591, "reward_total_mean": 0.8474001884460449, "reward_meter_mean": 0.9239417314529419, "reward_meter_std": 0.05428864806890488, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_total_composite_mean": 0.8474001884460449, "reward_total_composite_std": 0.15459680557250977} {"timestamp_utc": "2026-04-11T22:40:59Z", "mode": "train", "global_step": 684, "epoch": 0.027473189540908542, "loss": -0.0064, "grad_norm": 4.445869445800781, "learning_rate": 7.930303030303031e-06, "num_tokens": 1484327.0, "completions/mean_length": 80.25, "completions/min_length": 79.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.25, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.9331251978874207, "rewards/meter/std": 0.07388782501220703, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5, "rewards/repeat_penalty/std": 0.10690450668334961, "rewards/total_composite/mean": 0.46393126249313354, "rewards/total_composite/std": 0.09499123692512512, "reward": 0.46393126249313354, "reward_std": 0.09499124437570572, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01861550658941269, "sampling/sampling_logp_difference/max": 1.1865754127502441, "sampling/importance_sampling_ratio/min": 0.3052648901939392, "sampling/importance_sampling_ratio/mean": 1.0021919012069702, "sampling/importance_sampling_ratio/max": 1.7777718305587769, "entropy": 0.0625843945890665, "clip_ratio/low_mean": 0.007758883642964065, "clip_ratio/low_min": 0.007758883642964065, "clip_ratio/high_mean": 0.004629629664123058, "clip_ratio/high_max": 0.004629629664123058, "clip_ratio/region_mean": 0.012388513307087123, "reward_total_mean": 0.46393126249313354, "reward_meter_mean": 0.9331251978874207, "reward_meter_std": 0.07388782501220703, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5, "reward_repeat_penalty_std": 0.10690450668334961, "reward_total_composite_mean": 0.46393126249313354, "reward_total_composite_std": 0.09499123692512512} {"timestamp_utc": "2026-04-11T22:41:05Z", "mode": "train", "global_step": 685, "epoch": 0.027513355022693496, "loss": 0.0202, "grad_norm": 2.300926685333252, "learning_rate": 7.927272727272729e-06, "num_tokens": 1487252.0, "completions/mean_length": 170.625, "completions/min_length": 165.0, "completions/max_length": 178.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 170.625, "completions/min_terminated_length": 165.0, "completions/max_terminated_length": 178.0, "rewards/meter/mean": 0.8037871718406677, "rewards/meter/std": 0.32751503586769104, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4107142686843872, "rewards/repeat_penalty/std": 0.050507623702287674, "rewards/total_composite/mean": 0.34419798851013184, "rewards/total_composite/std": 0.14113976061344147, "reward": 0.34419798851013184, "reward_std": 0.14113974571228027, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010442078113555908, "sampling/sampling_logp_difference/max": 1.643384337425232, "sampling/importance_sampling_ratio/min": 0.19332465529441833, "sampling/importance_sampling_ratio/mean": 0.9983497262001038, "sampling/importance_sampling_ratio/max": 1.4589576721191406, "entropy": 0.046010758727788925, "clip_ratio/low_mean": 0.0007022471982054412, "clip_ratio/low_min": 0.0007022471982054412, "clip_ratio/high_mean": 0.010335821425542235, "clip_ratio/high_max": 0.010335821425542235, "clip_ratio/region_mean": 0.011038068623747677, "reward_total_mean": 0.34419798851013184, "reward_meter_mean": 0.8037871718406677, "reward_meter_std": 0.32751503586769104, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4107142686843872, "reward_repeat_penalty_std": 0.050507623702287674, "reward_total_composite_mean": 0.34419798851013184, "reward_total_composite_std": 0.14113976061344147} {"timestamp_utc": "2026-04-11T22:41:15Z", "mode": "train", "global_step": 686, "epoch": 0.02755352050447845, "loss": -0.0998, "grad_norm": 1.047049641609192, "learning_rate": 7.924242424242426e-06, "num_tokens": 1490154.0, "completions/mean_length": 251.75, "completions/min_length": 212.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 214.57144165039062, "completions/min_terminated_length": 212.0, "completions/max_terminated_length": 222.0, "rewards/meter/mean": 0.9798320531845093, "rewards/meter/std": 0.047762468457221985, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.2828427255153656, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.32499998807907104, "rewards/repeat_penalty/std": 0.27645719051361084, "rewards/total_composite/mean": 0.2209167182445526, "rewards/total_composite/std": 0.04932519793510437, "reward": 0.2209167182445526, "reward_std": 0.04932519420981407, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011242986656725407, "sampling/sampling_logp_difference/max": 2.929636001586914, "sampling/importance_sampling_ratio/min": 0.053416479378938675, "sampling/importance_sampling_ratio/mean": 0.9986944198608398, "sampling/importance_sampling_ratio/max": 1.7827941179275513, "entropy": 0.021744283847510815, "clip_ratio/low_mean": 0.003458057180978358, "clip_ratio/low_min": 0.003458057180978358, "clip_ratio/high_mean": 0.00234195904340595, "clip_ratio/high_max": 0.00234195904340595, "clip_ratio/region_mean": 0.005800016224384308, "reward_total_mean": 0.2209167182445526, "reward_meter_mean": 0.9798320531845093, "reward_meter_std": 0.047762468457221985, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.2828427255153656, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.32499998807907104, "reward_repeat_penalty_std": 0.27645719051361084, "reward_total_composite_mean": 0.2209167182445526, "reward_total_composite_std": 0.04932519793510437} {"timestamp_utc": "2026-04-11T22:41:19Z", "mode": "train", "global_step": 687, "epoch": 0.027593685986263404, "loss": 0.0292, "grad_norm": 10.655113220214844, "learning_rate": 7.921212121212122e-06, "num_tokens": 1491874.0, "completions/mean_length": 71.0, "completions/min_length": 67.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.8845878839492798, "rewards/meter/std": 0.2909509837627411, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.8430095314979553, "rewards/total_composite/std": 0.2961682975292206, "reward": 0.8430095314979553, "reward_std": 0.2961682677268982, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05931292846798897, "sampling/sampling_logp_difference/max": 2.937185287475586, "sampling/importance_sampling_ratio/min": 0.05301474407315254, "sampling/importance_sampling_ratio/mean": 1.0036027431488037, "sampling/importance_sampling_ratio/max": 1.9307304620742798, "entropy": 0.1976525131613016, "clip_ratio/low_mean": 0.011955027701333165, "clip_ratio/low_min": 0.011955027701333165, "clip_ratio/high_mean": 0.020970338257029653, "clip_ratio/high_max": 0.020970338257029653, "clip_ratio/region_mean": 0.03292536595836282, "reward_total_mean": 0.8430095314979553, "reward_meter_mean": 0.8845878839492798, "reward_meter_std": 0.2909509837627411, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.8430095314979553, "reward_total_composite_std": 0.2961682975292206} {"timestamp_utc": "2026-04-11T22:41:24Z", "mode": "train", "global_step": 688, "epoch": 0.027633851468048358, "loss": -0.0086, "grad_norm": 1.4717203378677368, "learning_rate": 7.918181818181819e-06, "num_tokens": 1493736.0, "completions/mean_length": 67.75, "completions/min_length": 66.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.75, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9984875321388245, "rewards/meter/std": 0.0001689638738753274, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6656583547592163, "rewards/total_composite/std": 0.00011263116175541654, "reward": 0.6656583547592163, "reward_std": 0.00011264239583397284, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018908310681581497, "sampling/sampling_logp_difference/max": 2.604221820831299, "sampling/importance_sampling_ratio/min": 0.07396066933870316, "sampling/importance_sampling_ratio/mean": 0.9985640048980713, "sampling/importance_sampling_ratio/max": 1.6862038373947144, "entropy": 0.06369170360267162, "clip_ratio/low_mean": 0.011363636702299118, "clip_ratio/low_min": 0.011363636702299118, "clip_ratio/high_mean": 0.010716472752392292, "clip_ratio/high_max": 0.010716472752392292, "clip_ratio/region_mean": 0.02208010945469141, "reward_total_mean": 0.6656583547592163, "reward_meter_mean": 0.9984875321388245, "reward_meter_std": 0.0001689638738753274, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6656583547592163, "reward_total_composite_std": 0.00011263116175541654} {"timestamp_utc": "2026-04-11T22:41:30Z", "mode": "train", "global_step": 689, "epoch": 0.027674016949833312, "loss": 0.0309, "grad_norm": 2.0693576335906982, "learning_rate": 7.915151515151516e-06, "num_tokens": 1497021.0, "completions/mean_length": 202.625, "completions/min_length": 191.0, "completions/max_length": 212.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 202.625, "completions/min_terminated_length": 191.0, "completions/max_terminated_length": 212.0, "rewards/meter/mean": 0.995306134223938, "rewards/meter/std": 0.0053658634424209595, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4194444417953491, "rewards/repeat_penalty/std": 0.13975918292999268, "rewards/total_composite/mean": 0.4050329029560089, "rewards/total_composite/std": 0.1355670541524887, "reward": 0.4050329029560089, "reward_std": 0.1355670541524887, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014271133579313755, "sampling/sampling_logp_difference/max": 2.4142937660217285, "sampling/importance_sampling_ratio/min": 0.08943047374486923, "sampling/importance_sampling_ratio/mean": 0.9986175298690796, "sampling/importance_sampling_ratio/max": 1.674434781074524, "entropy": 0.03671248443424702, "clip_ratio/low_mean": 0.007414351915940642, "clip_ratio/low_min": 0.007414351915940642, "clip_ratio/high_mean": 0.0019430051324889064, "clip_ratio/high_max": 0.0019430051324889064, "clip_ratio/region_mean": 0.009357357048429549, "reward_total_mean": 0.4050329029560089, "reward_meter_mean": 0.995306134223938, "reward_meter_std": 0.0053658634424209595, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4194444417953491, "reward_repeat_penalty_std": 0.13975918292999268, "reward_total_composite_mean": 0.4050329029560089, "reward_total_composite_std": 0.1355670541524887} {"timestamp_utc": "2026-04-11T22:41:39Z", "mode": "train", "global_step": 690, "epoch": 0.027714182431618266, "loss": 0.0024, "grad_norm": 1.1219055652618408, "learning_rate": 7.912121212121213e-06, "num_tokens": 1502074.0, "completions/mean_length": 401.625, "completions/min_length": 400.0, "completions/max_length": 402.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 401.625, "completions/min_terminated_length": 400.0, "completions/max_terminated_length": 402.0, "rewards/meter/mean": 0.9981971979141235, "rewards/meter/std": 0.000627980858553201, "rewards/count_adherence/mean": 0.7142857313156128, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.2503289580345154, "rewards/repeat_penalty/std": 0.23873279988765717, "rewards/total_composite/mean": 0.1784874051809311, "rewards/total_composite/std": 0.17026527225971222, "reward": 0.1784874051809311, "reward_std": 0.17026525735855103, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005141077097505331, "sampling/sampling_logp_difference/max": 1.3938775062561035, "sampling/importance_sampling_ratio/min": 0.24811138212680817, "sampling/importance_sampling_ratio/mean": 1.0005240440368652, "sampling/importance_sampling_ratio/max": 1.6704438924789429, "entropy": 0.027160495053976774, "clip_ratio/low_mean": 0.0021797263179905713, "clip_ratio/low_min": 0.0021797263179905713, "clip_ratio/high_mean": 0.0015578281017951667, "clip_ratio/high_max": 0.0015578281017951667, "clip_ratio/region_mean": 0.003737554419785738, "reward_total_mean": 0.1784874051809311, "reward_meter_mean": 0.9981971979141235, "reward_meter_std": 0.000627980858553201, "reward_count_adherence_mean": 0.7142857313156128, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.2503289580345154, "reward_repeat_penalty_std": 0.23873279988765717, "reward_total_composite_mean": 0.1784874051809311, "reward_total_composite_std": 0.17026527225971222} {"timestamp_utc": "2026-04-11T22:41:43Z", "mode": "train", "global_step": 691, "epoch": 0.02775434791340322, "loss": -0.1031, "grad_norm": 11.69098949432373, "learning_rate": 7.909090909090909e-06, "num_tokens": 1503707.0, "completions/mean_length": 38.125, "completions/min_length": 19.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.125, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.4370739161968231, "rewards/meter/std": 0.2700711488723755, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4370739161968231, "rewards/total_composite/std": 0.2700711488723755, "reward": 0.4370739161968231, "reward_std": 0.2700711488723755, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0835486352443695, "sampling/sampling_logp_difference/max": 1.102844476699829, "sampling/importance_sampling_ratio/min": 0.3319256007671356, "sampling/importance_sampling_ratio/mean": 1.0089696645736694, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5944820679724216, "clip_ratio/low_mean": 0.019354344811290503, "clip_ratio/low_min": 0.019354344811290503, "clip_ratio/high_mean": 0.04530784301459789, "clip_ratio/high_max": 0.04530784301459789, "clip_ratio/region_mean": 0.0646621878258884, "reward_total_mean": 0.4370739161968231, "reward_meter_mean": 0.4370739161968231, "reward_meter_std": 0.2700711488723755, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.4370739161968231, "reward_total_composite_std": 0.2700711488723755} {"timestamp_utc": "2026-04-11T22:41:47Z", "mode": "train", "global_step": 692, "epoch": 0.027794513395188174, "loss": 0.0153, "grad_norm": 5.031876564025879, "learning_rate": 7.906060606060608e-06, "num_tokens": 1505712.0, "completions/mean_length": 86.625, "completions/min_length": 84.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.625, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.9942313432693481, "rewards/meter/std": 0.0012215422466397285, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9942313432693481, "rewards/total_composite/std": 0.0012215422466397285, "reward": 0.9942313432693481, "reward_std": 0.00122154806740582, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005889675114303827, "sampling/sampling_logp_difference/max": 0.44769424200057983, "sampling/importance_sampling_ratio/min": 0.7906866669654846, "sampling/importance_sampling_ratio/mean": 1.003043293952942, "sampling/importance_sampling_ratio/max": 1.5647001266479492, "entropy": 0.04205932654440403, "clip_ratio/low_mean": 0.0014367816038429737, "clip_ratio/low_min": 0.0014367816038429737, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0014367816038429737, "reward_total_mean": 0.9942313432693481, "reward_meter_mean": 0.9942313432693481, "reward_meter_std": 0.0012215422466397285, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9942313432693481, "reward_total_composite_std": 0.0012215422466397285} {"timestamp_utc": "2026-04-11T22:41:52Z", "mode": "train", "global_step": 693, "epoch": 0.027834678876973128, "loss": 0.0041, "grad_norm": 10.364151954650879, "learning_rate": 7.903030303030303e-06, "num_tokens": 1507385.0, "completions/mean_length": 54.125, "completions/min_length": 52.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.125, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9447178244590759, "rewards/meter/std": 0.019572317600250244, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9447178244590759, "rewards/total_composite/std": 0.019572317600250244, "reward": 0.9447178244590759, "reward_std": 0.01957232505083084, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024356268346309662, "sampling/sampling_logp_difference/max": 1.9013309478759766, "sampling/importance_sampling_ratio/min": 0.14936968684196472, "sampling/importance_sampling_ratio/mean": 1.003846287727356, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0895584006793797, "clip_ratio/low_mean": 0.009437322150915861, "clip_ratio/low_min": 0.009437322150915861, "clip_ratio/high_mean": 0.01416083937510848, "clip_ratio/high_max": 0.01416083937510848, "clip_ratio/region_mean": 0.02359816152602434, "reward_total_mean": 0.9447178244590759, "reward_meter_mean": 0.9447178244590759, "reward_meter_std": 0.019572317600250244, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9447178244590759, "reward_total_composite_std": 0.019572317600250244} {"timestamp_utc": "2026-04-11T22:41:56Z", "mode": "train", "global_step": 694, "epoch": 0.02787484435875808, "loss": 0.0035, "grad_norm": 6.670380115509033, "learning_rate": 7.9e-06, "num_tokens": 1509273.0, "completions/mean_length": 74.0, "completions/min_length": 70.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.0, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.610145092010498, "rewards/meter/std": 0.23973627388477325, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.5444035530090332, "rewards/total_composite/std": 0.24921227991580963, "reward": 0.5444035530090332, "reward_std": 0.24921227991580963, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030294643715023994, "sampling/sampling_logp_difference/max": 1.8091429471969604, "sampling/importance_sampling_ratio/min": 0.16379445791244507, "sampling/importance_sampling_ratio/mean": 1.0047415494918823, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17548873648047447, "clip_ratio/low_mean": 0.01502489356789738, "clip_ratio/low_min": 0.01502489356789738, "clip_ratio/high_mean": 0.02390445303171873, "clip_ratio/high_max": 0.02390445303171873, "clip_ratio/region_mean": 0.03892934659961611, "reward_total_mean": 0.5444035530090332, "reward_meter_mean": 0.610145092010498, "reward_meter_std": 0.23973627388477325, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.17251639068126678, "reward_total_composite_mean": 0.5444035530090332, "reward_total_composite_std": 0.24921227991580963} {"timestamp_utc": "2026-04-11T22:42:01Z", "mode": "train", "global_step": 695, "epoch": 0.02791500984054304, "loss": -0.0124, "grad_norm": 3.7013967037200928, "learning_rate": 7.896969696969698e-06, "num_tokens": 1511199.0, "completions/mean_length": 80.75, "completions/min_length": 78.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.75, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9964468479156494, "rewards/meter/std": 0.0019429969834163785, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7916666865348816, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.788793683052063, "rewards/total_composite/std": 0.17156179249286652, "reward": 0.788793683052063, "reward_std": 0.17156179249286652, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012971381656825542, "sampling/sampling_logp_difference/max": 1.7286226749420166, "sampling/importance_sampling_ratio/min": 0.1775287538766861, "sampling/importance_sampling_ratio/mean": 1.0024663209915161, "sampling/importance_sampling_ratio/max": 1.8497363328933716, "entropy": 0.06276643788442016, "clip_ratio/low_mean": 0.0062915480230003595, "clip_ratio/low_min": 0.0062915480230003595, "clip_ratio/high_mean": 0.004360056365840137, "clip_ratio/high_max": 0.004360056365840137, "clip_ratio/region_mean": 0.010651604388840497, "reward_total_mean": 0.788793683052063, "reward_meter_mean": 0.9964468479156494, "reward_meter_std": 0.0019429969834163785, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7916666865348816, "reward_repeat_penalty_std": 0.17251639068126678, "reward_total_composite_mean": 0.788793683052063, "reward_total_composite_std": 0.17156179249286652} {"timestamp_utc": "2026-04-11T22:42:06Z", "mode": "train", "global_step": 696, "epoch": 0.027955175322327993, "loss": -0.0063, "grad_norm": 3.119380235671997, "learning_rate": 7.893939393939395e-06, "num_tokens": 1513573.0, "completions/mean_length": 125.75, "completions/min_length": 112.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 125.75, "completions/min_terminated_length": 112.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9228010177612305, "rewards/meter/std": 0.20124337077140808, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7000000476837158, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.6532999277114868, "rewards/total_composite/std": 0.18960076570510864, "reward": 0.6532999277114868, "reward_std": 0.18960076570510864, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016659488901495934, "sampling/sampling_logp_difference/max": 1.3884687423706055, "sampling/importance_sampling_ratio/min": 0.2494570016860962, "sampling/importance_sampling_ratio/mean": 0.9984103441238403, "sampling/importance_sampling_ratio/max": 1.9363112449645996, "entropy": 0.07676348416134715, "clip_ratio/low_mean": 0.0020850637229159474, "clip_ratio/low_min": 0.0020850637229159474, "clip_ratio/high_mean": 0.00876707280986011, "clip_ratio/high_max": 0.00876707280986011, "clip_ratio/region_mean": 0.010852136532776058, "reward_total_mean": 0.6532999277114868, "reward_meter_mean": 0.9228010177612305, "reward_meter_std": 0.20124337077140808, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7000000476837158, "reward_repeat_penalty_std": 0.10690449178218842, "reward_total_composite_mean": 0.6532999277114868, "reward_total_composite_std": 0.18960076570510864} {"timestamp_utc": "2026-04-11T22:42:11Z", "mode": "train", "global_step": 697, "epoch": 0.027995340804112947, "loss": 0.0606, "grad_norm": 4.277256011962891, "learning_rate": 7.89090909090909e-06, "num_tokens": 1515360.0, "completions/mean_length": 59.375, "completions/min_length": 55.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.375, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9654116630554199, "rewards/meter/std": 0.034304678440093994, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.8894380331039429, "rewards/total_composite/std": 0.17405925691127777, "reward": 0.8894380331039429, "reward_std": 0.17405925691127777, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020010532811284065, "sampling/sampling_logp_difference/max": 1.0107874870300293, "sampling/importance_sampling_ratio/min": 0.3639322817325592, "sampling/importance_sampling_ratio/mean": 0.99860018491745, "sampling/importance_sampling_ratio/max": 1.4003076553344727, "entropy": 0.09744885191321373, "clip_ratio/low_mean": 0.003846153849735856, "clip_ratio/low_min": 0.003846153849735856, "clip_ratio/high_mean": 0.01933896285481751, "clip_ratio/high_max": 0.01933896285481751, "clip_ratio/region_mean": 0.023185116704553366, "reward_total_mean": 0.8894380331039429, "reward_meter_mean": 0.9654116630554199, "reward_meter_std": 0.034304678440093994, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_total_composite_mean": 0.8894380331039429, "reward_total_composite_std": 0.17405925691127777} {"timestamp_utc": "2026-04-11T22:42:21Z", "mode": "train", "global_step": 698, "epoch": 0.0280355062858979, "loss": -0.0873, "grad_norm": 2.2967092990875244, "learning_rate": 7.88787878787879e-06, "num_tokens": 1519450.0, "completions/mean_length": 382.25, "completions/min_length": 321.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 363.71429443359375, "completions/min_terminated_length": 321.0, "completions/max_terminated_length": 412.0, "rewards/meter/mean": 0.6105729937553406, "rewards/meter/std": 0.4790833294391632, "rewards/count_adherence/mean": 0.8854166865348816, "rewards/count_adherence/std": 0.1254950612783432, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 0.3964124917984009, "rewards/repeat_penalty/std": 0.25513792037963867, "rewards/total_composite/mean": 0.18387337028980255, "rewards/total_composite/std": 0.21844197809696198, "reward": 0.18387337028980255, "reward_std": 0.21844197809696198, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0396236851811409, "sampling/sampling_logp_difference/max": 6.693481922149658, "sampling/importance_sampling_ratio/min": 0.001238961354829371, "sampling/importance_sampling_ratio/mean": 0.9987672567367554, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.18515709601342678, "clip_ratio/low_mean": 0.007161757908761501, "clip_ratio/low_min": 0.007161757908761501, "clip_ratio/high_mean": 0.015907755587249994, "clip_ratio/high_max": 0.015907755587249994, "clip_ratio/region_mean": 0.023069513496011496, "reward_total_mean": 0.18387337028980255, "reward_meter_mean": 0.6105729937553406, "reward_meter_std": 0.4790833294391632, "reward_count_adherence_mean": 0.8854166865348816, "reward_count_adherence_std": 0.1254950612783432, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 0.3964124917984009, "reward_repeat_penalty_std": 0.25513792037963867, "reward_total_composite_mean": 0.18387337028980255, "reward_total_composite_std": 0.21844197809696198} {"timestamp_utc": "2026-04-11T22:42:25Z", "mode": "train", "global_step": 699, "epoch": 0.028075671767682855, "loss": 0.0056, "grad_norm": 8.473615646362305, "learning_rate": 7.884848484848485e-06, "num_tokens": 1521331.0, "completions/mean_length": 67.125, "completions/min_length": 63.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.125, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9927859306335449, "rewards/meter/std": 0.014541360549628735, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8333333730697632, "rewards/repeat_penalty/std": 0.17817415297031403, "rewards/total_composite/mean": 0.8282062411308289, "rewards/total_composite/std": 0.18183693289756775, "reward": 0.8282062411308289, "reward_std": 0.18183691799640656, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03374454379081726, "sampling/sampling_logp_difference/max": 1.766160249710083, "sampling/importance_sampling_ratio/min": 0.1795266717672348, "sampling/importance_sampling_ratio/mean": 0.9997567534446716, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.14588068891316652, "clip_ratio/low_mean": 0.00747219193726778, "clip_ratio/low_min": 0.00747219193726778, "clip_ratio/high_mean": 0.02036348171532154, "clip_ratio/high_max": 0.02036348171532154, "clip_ratio/region_mean": 0.02783567365258932, "reward_total_mean": 0.8282062411308289, "reward_meter_mean": 0.9927859306335449, "reward_meter_std": 0.014541360549628735, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8333333730697632, "reward_repeat_penalty_std": 0.17817415297031403, "reward_total_composite_mean": 0.8282062411308289, "reward_total_composite_std": 0.18183693289756775} {"timestamp_utc": "2026-04-11T22:42:35Z", "mode": "train", "global_step": 700, "epoch": 0.02811583724946781, "loss": 0.0097, "grad_norm": 0.8298696279525757, "learning_rate": 7.881818181818182e-06, "num_tokens": 1526242.0, "completions/mean_length": 414.875, "completions/min_length": 401.0, "completions/max_length": 444.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 414.875, "completions/min_terminated_length": 401.0, "completions/max_terminated_length": 444.0, "rewards/meter/mean": 0.9960967302322388, "rewards/meter/std": 0.0013218529056757689, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.029462797567248344, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.19121241569519043, "rewards/repeat_penalty/std": 0.07494427263736725, "rewards/total_composite/mean": 0.16019713878631592, "rewards/total_composite/std": 0.061176449060440063, "reward": 0.16019713878631592, "reward_std": 0.061176449060440063, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008506695739924908, "sampling/sampling_logp_difference/max": 1.432920217514038, "sampling/importance_sampling_ratio/min": 0.238611102104187, "sampling/importance_sampling_ratio/mean": 1.0012154579162598, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.045434954110533, "clip_ratio/low_mean": 0.00364298140630126, "clip_ratio/low_min": 0.00364298140630126, "clip_ratio/high_mean": 0.006157157360576093, "clip_ratio/high_max": 0.006157157360576093, "clip_ratio/region_mean": 0.009800138766877353, "reward_total_mean": 0.16019713878631592, "reward_meter_mean": 0.9960967302322388, "reward_meter_std": 0.0013218529056757689, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.029462797567248344, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.19121241569519043, "reward_repeat_penalty_std": 0.07494427263736725, "reward_total_composite_mean": 0.16019713878631592, "reward_total_composite_std": 0.061176449060440063} {"timestamp_utc": "2026-04-11T22:44:02Z", "mode": "eval", "global_step": 700, "epoch": 0.02811583724946781, "eval_loss": NaN, "eval_runtime": 87.7418, "eval_samples_per_second": 1.185, "eval_steps_per_second": 0.148, "eval_num_tokens": 1526242.0, "eval_completions/mean_length": 249.43269230769232, "eval_completions/min_length": 68.53846153846153, "eval_completions/max_length": 471.9230769230769, "eval_completions/clipped_ratio": 0.11538461538461539, "eval_completions/mean_terminated_length": 217.6739994929387, "eval_completions/min_terminated_length": 68.53846153846153, "eval_completions/max_terminated_length": 415.15384615384613, "eval_rewards/meter/mean": 0.6321157022164419, "eval_rewards/meter/std": 0.37711624113413006, "eval_rewards/count_adherence/mean": 0.8881359283740704, "eval_rewards/count_adherence/std": 0.16248861929545036, "eval_rewards/arabic_clean/mean": 0.9038461538461539, "eval_rewards/arabic_clean/std": 0.2343954168833219, "eval_rewards/repeat_penalty/mean": 0.5953293947073129, "eval_rewards/repeat_penalty/std": 0.32899803152451146, "eval_rewards/total_composite/mean": 0.29438196466519284, "eval_rewards/total_composite/std": 0.28008361991781455, "eval_reward": 0.29438196466519284, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.012457216982371531, "eval_sampling/sampling_logp_difference/max": 0.881988103573139, "eval_sampling/importance_sampling_ratio/min": 0.4343368663237645, "eval_sampling/importance_sampling_ratio/mean": 1.0041690973135142, "eval_sampling/importance_sampling_ratio/max": 1.455959943624643, "eval_entropy": 0.1376804428604933, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.29438196466519284, "eval_reward_meter_mean": 0.6321157022164419, "eval_reward_meter_std": 0.37711624113413006, "eval_reward_count_adherence_mean": 0.8881359283740704, "eval_reward_count_adherence_std": 0.16248861929545036, "eval_reward_arabic_clean_mean": 0.9038461538461539, "eval_reward_arabic_clean_std": 0.2343954168833219, "eval_reward_repeat_penalty_mean": 0.5953293947073129, "eval_reward_repeat_penalty_std": 0.32899803152451146, "eval_reward_total_composite_mean": 0.29438196466519284, "eval_reward_total_composite_std": 0.28008361991781455} {"timestamp_utc": "2026-04-11T22:44:17Z", "mode": "train", "global_step": 701, "epoch": 0.028156002731252763, "loss": -0.12, "grad_norm": 2.5312631130218506, "learning_rate": 7.87878787878788e-06, "num_tokens": 1528168.0, "completions/mean_length": 138.75, "completions/min_length": 79.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 85.42857360839844, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.646858811378479, "rewards/meter/std": 0.40135496854782104, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6387161016464233, "rewards/total_composite/std": 0.41526278853416443, "reward": 0.6387161016464233, "reward_std": 0.41526278853416443, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017966095358133316, "sampling/sampling_logp_difference/max": 0.638648509979248, "sampling/importance_sampling_ratio/min": 0.5280055403709412, "sampling/importance_sampling_ratio/mean": 1.0036530494689941, "sampling/importance_sampling_ratio/max": 1.6849335432052612, "entropy": 0.11331802047789097, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.015958538744598627, "clip_ratio/high_max": 0.015958538744598627, "clip_ratio/region_mean": 0.015958538744598627, "reward_total_mean": 0.6387161016464233, "reward_meter_mean": 0.646858811378479, "reward_meter_std": 0.40135496854782104, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6387161016464233, "reward_total_composite_std": 0.41526278853416443} {"timestamp_utc": "2026-04-11T22:44:27Z", "mode": "train", "global_step": 702, "epoch": 0.028196168213037717, "loss": -0.0691, "grad_norm": 3.0773770809173584, "learning_rate": 7.875757575757577e-06, "num_tokens": 1529770.0, "completions/mean_length": 128.25, "completions/min_length": 71.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 73.42857360839844, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.3803873062133789, "rewards/meter/std": 0.26946189999580383, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.33759811520576477, "rewards/total_composite/std": 0.30163025856018066, "reward": 0.33759811520576477, "reward_std": 0.3016302287578583, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04749147966504097, "sampling/sampling_logp_difference/max": 1.9366450309753418, "sampling/importance_sampling_ratio/min": 0.14418688416481018, "sampling/importance_sampling_ratio/mean": 1.011560082435608, "sampling/importance_sampling_ratio/max": 1.7729606628417969, "entropy": 0.37723080068826675, "clip_ratio/low_mean": 0.01317842910066247, "clip_ratio/low_min": 0.01317842910066247, "clip_ratio/high_mean": 0.0155344782397151, "clip_ratio/high_max": 0.0155344782397151, "clip_ratio/region_mean": 0.02871290734037757, "reward_total_mean": 0.33759811520576477, "reward_meter_mean": 0.3803873062133789, "reward_meter_std": 0.26946189999580383, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.33759811520576477, "reward_total_composite_std": 0.30163025856018066} {"timestamp_utc": "2026-04-11T22:44:34Z", "mode": "train", "global_step": 703, "epoch": 0.02823633369482267, "loss": -0.0245, "grad_norm": 1.2964930534362793, "learning_rate": 7.872727272727273e-06, "num_tokens": 1533320.0, "completions/mean_length": 229.75, "completions/min_length": 206.0, "completions/max_length": 242.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 229.75, "completions/min_terminated_length": 206.0, "completions/max_terminated_length": 242.0, "rewards/meter/mean": 0.9970081448554993, "rewards/meter/std": 0.001187506248243153, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.2613636255264282, "rewards/repeat_penalty/std": 0.197011336684227, "rewards/total_composite/mean": 0.2607119679450989, "rewards/total_composite/std": 0.19680176675319672, "reward": 0.2607119679450989, "reward_std": 0.1968017816543579, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014087834395468235, "sampling/sampling_logp_difference/max": 4.6370849609375, "sampling/importance_sampling_ratio/min": 0.009685891680419445, "sampling/importance_sampling_ratio/mean": 0.9988393187522888, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03166929807048291, "clip_ratio/low_mean": 0.0028346364269964397, "clip_ratio/low_min": 0.0028346364269964397, "clip_ratio/high_mean": 0.004737977171316743, "clip_ratio/high_max": 0.004737977171316743, "clip_ratio/region_mean": 0.007572613598313183, "reward_total_mean": 0.2607119679450989, "reward_meter_mean": 0.9970081448554993, "reward_meter_std": 0.001187506248243153, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.2613636255264282, "reward_repeat_penalty_std": 0.197011336684227, "reward_total_composite_mean": 0.2607119679450989, "reward_total_composite_std": 0.19680176675319672} {"timestamp_utc": "2026-04-11T22:44:44Z", "mode": "train", "global_step": 704, "epoch": 0.028276499176607624, "loss": -0.1074, "grad_norm": 3.015087127685547, "learning_rate": 7.86969696969697e-06, "num_tokens": 1536668.0, "completions/mean_length": 364.5, "completions/min_length": 229.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 315.3333435058594, "completions/min_terminated_length": 229.0, "completions/max_terminated_length": 368.0, "rewards/meter/mean": 0.20065514743328094, "rewards/meter/std": 0.3053584098815918, "rewards/count_adherence/mean": 0.8522727489471436, "rewards/count_adherence/std": 0.21697448194026947, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 0.6299689412117004, "rewards/repeat_penalty/std": 0.28235092759132385, "rewards/total_composite/mean": 0.0969100296497345, "rewards/total_composite/std": 0.21232596039772034, "reward": 0.0969100296497345, "reward_std": 0.21232594549655914, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04316805675625801, "sampling/sampling_logp_difference/max": 2.4648847579956055, "sampling/importance_sampling_ratio/min": 0.08501863479614258, "sampling/importance_sampling_ratio/mean": 0.9969695806503296, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2017015889286995, "clip_ratio/low_mean": 0.01385743310675025, "clip_ratio/low_min": 0.01385743310675025, "clip_ratio/high_mean": 0.009915342554450035, "clip_ratio/high_max": 0.009915342554450035, "clip_ratio/region_mean": 0.023772775661200285, "reward_total_mean": 0.0969100296497345, "reward_meter_mean": 0.20065514743328094, "reward_meter_std": 0.3053584098815918, "reward_count_adherence_mean": 0.8522727489471436, "reward_count_adherence_std": 0.21697448194026947, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 0.6299689412117004, "reward_repeat_penalty_std": 0.28235092759132385, "reward_total_composite_mean": 0.0969100296497345, "reward_total_composite_std": 0.21232596039772034} {"timestamp_utc": "2026-04-11T22:44:49Z", "mode": "train", "global_step": 705, "epoch": 0.02831666465839258, "loss": -0.0075, "grad_norm": 2.615363836288452, "learning_rate": 7.866666666666667e-06, "num_tokens": 1538402.0, "completions/mean_length": 58.75, "completions/min_length": 56.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.75, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9921605587005615, "rewards/meter/std": 0.006017809733748436, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.909184455871582, "rewards/total_composite/std": 0.15155236423015594, "reward": 0.909184455871582, "reward_std": 0.15155236423015594, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021948276087641716, "sampling/sampling_logp_difference/max": 3.973776340484619, "sampling/importance_sampling_ratio/min": 0.01880229450762272, "sampling/importance_sampling_ratio/mean": 0.9994084239006042, "sampling/importance_sampling_ratio/max": 1.5749088525772095, "entropy": 0.053199955029413104, "clip_ratio/low_mean": 0.006398809840902686, "clip_ratio/low_min": 0.006398809840902686, "clip_ratio/high_mean": 0.006355932215228677, "clip_ratio/high_max": 0.006355932215228677, "clip_ratio/region_mean": 0.012754742056131363, "reward_total_mean": 0.909184455871582, "reward_meter_mean": 0.9921605587005615, "reward_meter_std": 0.006017809733748436, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_total_composite_mean": 0.909184455871582, "reward_total_composite_std": 0.15155236423015594} {"timestamp_utc": "2026-04-11T22:44:53Z", "mode": "train", "global_step": 706, "epoch": 0.028356830140177532, "loss": 0.0093, "grad_norm": 4.515374183654785, "learning_rate": 7.863636363636364e-06, "num_tokens": 1539816.0, "completions/mean_length": 44.75, "completions/min_length": 43.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.75, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.9157871603965759, "rewards/meter/std": 0.05085242539644241, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9157871603965759, "rewards/total_composite/std": 0.05085242539644241, "reward": 0.9157871603965759, "reward_std": 0.05085243284702301, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02820507250726223, "sampling/sampling_logp_difference/max": 0.7927889823913574, "sampling/importance_sampling_ratio/min": 0.4525808095932007, "sampling/importance_sampling_ratio/mean": 0.997020423412323, "sampling/importance_sampling_ratio/max": 1.2550123929977417, "entropy": 0.13661748263984919, "clip_ratio/low_mean": 0.0027777778450399637, "clip_ratio/low_min": 0.0027777778450399637, "clip_ratio/high_mean": 0.01414728700183332, "clip_ratio/high_max": 0.01414728700183332, "clip_ratio/region_mean": 0.016925064846873283, "reward_total_mean": 0.9157871603965759, "reward_meter_mean": 0.9157871603965759, "reward_meter_std": 0.05085242539644241, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9157871603965759, "reward_total_composite_std": 0.05085242539644241} {"timestamp_utc": "2026-04-11T22:44:59Z", "mode": "train", "global_step": 707, "epoch": 0.028396995621962486, "loss": -0.0003, "grad_norm": 2.2196364402770996, "learning_rate": 7.860606060606062e-06, "num_tokens": 1541901.0, "completions/mean_length": 84.625, "completions/min_length": 83.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.625, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.9956607818603516, "rewards/meter/std": 0.0007934165187180042, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9956607818603516, "rewards/total_composite/std": 0.0007934165187180042, "reward": 0.9956607818603516, "reward_std": 0.0007934237364679575, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026080820709466934, "sampling/sampling_logp_difference/max": 0.8360247611999512, "sampling/importance_sampling_ratio/min": 0.4566366374492645, "sampling/importance_sampling_ratio/mean": 1.0091147422790527, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15222040470689535, "clip_ratio/low_mean": 0.007405462441965938, "clip_ratio/low_min": 0.007405462441965938, "clip_ratio/high_mean": 0.013377037481404841, "clip_ratio/high_max": 0.013377037481404841, "clip_ratio/region_mean": 0.02078249992337078, "reward_total_mean": 0.9956607818603516, "reward_meter_mean": 0.9956607818603516, "reward_meter_std": 0.0007934165187180042, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9956607818603516, "reward_total_composite_std": 0.0007934165187180042} {"timestamp_utc": "2026-04-11T22:45:07Z", "mode": "train", "global_step": 708, "epoch": 0.02843716110374744, "loss": 0.0303, "grad_norm": 2.581402063369751, "learning_rate": 7.857575757575759e-06, "num_tokens": 1546099.0, "completions/mean_length": 330.75, "completions/min_length": 284.0, "completions/max_length": 358.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 330.75, "completions/min_terminated_length": 284.0, "completions/max_terminated_length": 358.0, "rewards/meter/mean": 0.5674466490745544, "rewards/meter/std": 0.4153376519680023, "rewards/count_adherence/mean": 0.9027777910232544, "rewards/count_adherence/std": 0.03928370773792267, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.34049707651138306, "rewards/repeat_penalty/std": 0.22640277445316315, "rewards/total_composite/mean": 0.2035118043422699, "rewards/total_composite/std": 0.23429380357265472, "reward": 0.2035118043422699, "reward_std": 0.23429378867149353, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023940538987517357, "sampling/sampling_logp_difference/max": 3.625192880630493, "sampling/importance_sampling_ratio/min": 0.026643957942724228, "sampling/importance_sampling_ratio/mean": 1.002025842666626, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10461874585598707, "clip_ratio/low_mean": 0.012675938894972205, "clip_ratio/low_min": 0.012675938894972205, "clip_ratio/high_mean": 0.010587403550744057, "clip_ratio/high_max": 0.010587403550744057, "clip_ratio/region_mean": 0.023263342445716262, "reward_total_mean": 0.2035118043422699, "reward_meter_mean": 0.5674466490745544, "reward_meter_std": 0.4153376519680023, "reward_count_adherence_mean": 0.9027777910232544, "reward_count_adherence_std": 0.03928370773792267, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.34049707651138306, "reward_repeat_penalty_std": 0.22640277445316315, "reward_total_composite_mean": 0.2035118043422699, "reward_total_composite_std": 0.23429380357265472} {"timestamp_utc": "2026-04-11T22:45:17Z", "mode": "train", "global_step": 709, "epoch": 0.028477326585532394, "loss": -0.3582, "grad_norm": 0.8571996688842773, "learning_rate": 7.854545454545454e-06, "num_tokens": 1549805.0, "completions/mean_length": 452.25, "completions/min_length": 372.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 416.3999938964844, "completions/min_terminated_length": 372.0, "completions/max_terminated_length": 472.0, "rewards/meter/mean": 0.8675932884216309, "rewards/meter/std": 0.35074320435523987, "rewards/count_adherence/mean": 0.6590909361839294, "rewards/count_adherence/std": 0.39101481437683105, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/repeat_penalty/mean": 0.6495236158370972, "rewards/repeat_penalty/std": 0.30929210782051086, "rewards/total_composite/mean": 0.25732704997062683, "rewards/total_composite/std": 0.2531158924102783, "reward": 0.25732704997062683, "reward_std": 0.2531158924102783, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019949674606323242, "sampling/sampling_logp_difference/max": 2.8196310997009277, "sampling/importance_sampling_ratio/min": 0.05962793529033661, "sampling/importance_sampling_ratio/mean": 1.0035980939865112, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0702343238517642, "clip_ratio/low_mean": 0.0015743073308840394, "clip_ratio/low_min": 0.0015743073308840394, "clip_ratio/high_mean": 0.005005840037483722, "clip_ratio/high_max": 0.005005840037483722, "clip_ratio/region_mean": 0.006580147368367761, "reward_total_mean": 0.25732704997062683, "reward_meter_mean": 0.8675932884216309, "reward_meter_std": 0.35074320435523987, "reward_count_adherence_mean": 0.6590909361839294, "reward_count_adherence_std": 0.39101481437683105, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_repeat_penalty_mean": 0.6495236158370972, "reward_repeat_penalty_std": 0.30929210782051086, "reward_total_composite_mean": 0.25732704997062683, "reward_total_composite_std": 0.2531158924102783} {"timestamp_utc": "2026-04-11T22:45:28Z", "mode": "train", "global_step": 710, "epoch": 0.028517492067317348, "loss": -0.2247, "grad_norm": 0.6216875314712524, "learning_rate": 7.851515151515152e-06, "num_tokens": 1554988.0, "completions/mean_length": 474.875, "completions/min_length": 439.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 469.5714416503906, "completions/min_terminated_length": 439.0, "completions/max_terminated_length": 497.0, "rewards/meter/mean": 0.9897267818450928, "rewards/meter/std": 0.019744135439395905, "rewards/count_adherence/mean": 0.7767857313156128, "rewards/count_adherence/std": 0.20360276103019714, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.5224603414535522, "rewards/repeat_penalty/std": 0.23697951436042786, "rewards/total_composite/mean": 0.3333708643913269, "rewards/total_composite/std": 0.18023891746997833, "reward": 0.3333708643913269, "reward_std": 0.18023891746997833, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010584630072116852, "sampling/sampling_logp_difference/max": 2.4926586151123047, "sampling/importance_sampling_ratio/min": 0.08268983662128448, "sampling/importance_sampling_ratio/mean": 1.0017552375793457, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.05670151812955737, "clip_ratio/low_mean": 0.0021170872496441007, "clip_ratio/low_min": 0.0021170872496441007, "clip_ratio/high_mean": 0.004514876694884151, "clip_ratio/high_max": 0.004514876694884151, "clip_ratio/region_mean": 0.006631963944528252, "reward_total_mean": 0.3333708643913269, "reward_meter_mean": 0.9897267818450928, "reward_meter_std": 0.019744135439395905, "reward_count_adherence_mean": 0.7767857313156128, "reward_count_adherence_std": 0.20360276103019714, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.5224603414535522, "reward_repeat_penalty_std": 0.23697951436042786, "reward_total_composite_mean": 0.3333708643913269, "reward_total_composite_std": 0.18023891746997833} {"timestamp_utc": "2026-04-11T22:45:38Z", "mode": "train", "global_step": 711, "epoch": 0.028557657549102302, "loss": -0.0049, "grad_norm": 0.6046677231788635, "learning_rate": 7.848484848484849e-06, "num_tokens": 1560550.0, "completions/mean_length": 440.25, "completions/min_length": 436.0, "completions/max_length": 461.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 440.25, "completions/min_terminated_length": 436.0, "completions/max_terminated_length": 461.0, "rewards/meter/mean": 0.9979845285415649, "rewards/meter/std": 0.00018585202633403242, "rewards/count_adherence/mean": 0.7946428060531616, "rewards/count_adherence/std": 0.025253823027014732, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.0676877498626709, "rewards/repeat_penalty/std": 0.023803479969501495, "rewards/total_composite/mean": 0.053851306438446045, "rewards/total_composite/std": 0.019493911415338516, "reward": 0.053851306438446045, "reward_std": 0.019493909552693367, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003577793249860406, "sampling/sampling_logp_difference/max": 1.1886392831802368, "sampling/importance_sampling_ratio/min": 0.30463549494743347, "sampling/importance_sampling_ratio/mean": 1.0001322031021118, "sampling/importance_sampling_ratio/max": 1.460545539855957, "entropy": 0.011412000167183578, "clip_ratio/low_mean": 0.0008561643480788916, "clip_ratio/low_min": 0.0008561643480788916, "clip_ratio/high_mean": 0.0016890882980078459, "clip_ratio/high_max": 0.0016890882980078459, "clip_ratio/region_mean": 0.0025452526460867375, "reward_total_mean": 0.053851306438446045, "reward_meter_mean": 0.9979845285415649, "reward_meter_std": 0.00018585202633403242, "reward_count_adherence_mean": 0.7946428060531616, "reward_count_adherence_std": 0.025253823027014732, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.0676877498626709, "reward_repeat_penalty_std": 0.023803479969501495, "reward_total_composite_mean": 0.053851306438446045, "reward_total_composite_std": 0.019493911415338516} {"timestamp_utc": "2026-04-11T22:45:44Z", "mode": "train", "global_step": 712, "epoch": 0.028597823030887256, "loss": 0.0004, "grad_norm": 4.805282115936279, "learning_rate": 7.845454545454546e-06, "num_tokens": 1562410.0, "completions/mean_length": 69.5, "completions/min_length": 68.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.5, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.8851493000984192, "rewards/meter/std": 0.1646534502506256, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7916666865348816, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.7055873274803162, "rewards/total_composite/std": 0.21653573215007782, "reward": 0.7055873274803162, "reward_std": 0.216535747051239, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03851144388318062, "sampling/sampling_logp_difference/max": 1.9700713157653809, "sampling/importance_sampling_ratio/min": 0.1394468992948532, "sampling/importance_sampling_ratio/mean": 0.997905969619751, "sampling/importance_sampling_ratio/max": 1.8839155435562134, "entropy": 0.1954718241468072, "clip_ratio/low_mean": 0.016306631732732058, "clip_ratio/low_min": 0.016306631732732058, "clip_ratio/high_mean": 0.014235412469133735, "clip_ratio/high_max": 0.014235412469133735, "clip_ratio/region_mean": 0.030542044201865792, "reward_total_mean": 0.7055873274803162, "reward_meter_mean": 0.8851493000984192, "reward_meter_std": 0.1646534502506256, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7916666865348816, "reward_repeat_penalty_std": 0.17251639068126678, "reward_total_composite_mean": 0.7055873274803162, "reward_total_composite_std": 0.21653573215007782} {"timestamp_utc": "2026-04-11T22:45:54Z", "mode": "train", "global_step": 713, "epoch": 0.02863798851267221, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.842424242424243e-06, "num_tokens": 1564234.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9704335927963257, "rewards/meter/std": 0.07359233498573303, "rewards/count_adherence/mean": 0.8166667222976685, "rewards/count_adherence/std": 0.030860668048262596, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5789903998374939, "rewards/repeat_penalty/std": 0.045423366129398346, "rewards/total_composite/mean": 0.4566306471824646, "rewards/total_composite/std": 0.02297402359545231, "reward": 0.4566306471824646, "reward_std": 0.02297401800751686, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.4566306471824646, "reward_meter_mean": 0.9704335927963257, "reward_meter_std": 0.07359233498573303, "reward_count_adherence_mean": 0.8166667222976685, "reward_count_adherence_std": 0.030860668048262596, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5789903998374939, "reward_repeat_penalty_std": 0.045423366129398346, "reward_total_composite_mean": 0.4566306471824646, "reward_total_composite_std": 0.02297402359545231} {"timestamp_utc": "2026-04-11T22:46:02Z", "mode": "train", "global_step": 714, "epoch": 0.028678153994457164, "loss": 0.0251, "grad_norm": 0.7193350791931152, "learning_rate": 7.83939393939394e-06, "num_tokens": 1568398.0, "completions/mean_length": 336.5, "completions/min_length": 311.0, "completions/max_length": 345.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 336.5, "completions/min_terminated_length": 311.0, "completions/max_terminated_length": 345.0, "rewards/meter/mean": 0.9977849721908569, "rewards/meter/std": 0.0006831432110629976, "rewards/count_adherence/mean": 0.7142857313156128, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.19117647409439087, "rewards/repeat_penalty/std": 0.16262187063694, "rewards/total_composite/mean": 0.13619501888751984, "rewards/total_composite/std": 0.1156935766339302, "reward": 0.13619501888751984, "reward_std": 0.11569356918334961, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003576630027964711, "sampling/sampling_logp_difference/max": 1.290938377380371, "sampling/importance_sampling_ratio/min": 0.2750125825405121, "sampling/importance_sampling_ratio/mean": 0.9997386932373047, "sampling/importance_sampling_ratio/max": 1.3743659257888794, "entropy": 0.018028545891866088, "clip_ratio/low_mean": 0.0018522579048294574, "clip_ratio/low_min": 0.0018522579048294574, "clip_ratio/high_mean": 0.0012019231216982007, "clip_ratio/high_max": 0.0012019231216982007, "clip_ratio/region_mean": 0.003054181026527658, "reward_total_mean": 0.13619501888751984, "reward_meter_mean": 0.9977849721908569, "reward_meter_std": 0.0006831432110629976, "reward_count_adherence_mean": 0.7142857313156128, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.19117647409439087, "reward_repeat_penalty_std": 0.16262187063694, "reward_total_composite_mean": 0.13619501888751984, "reward_total_composite_std": 0.1156935766339302} {"timestamp_utc": "2026-04-11T22:46:08Z", "mode": "train", "global_step": 715, "epoch": 0.028718319476242118, "loss": 0.0035, "grad_norm": 4.87160062789917, "learning_rate": 7.836363636363638e-06, "num_tokens": 1570787.0, "completions/mean_length": 123.625, "completions/min_length": 115.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 123.625, "completions/min_terminated_length": 115.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.9901050329208374, "rewards/meter/std": 0.0037112515419721603, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.1608559489250183, "rewards/total_composite/mean": 0.7246866822242737, "rewards/total_composite/std": 0.15871340036392212, "reward": 0.7246866822242737, "reward_std": 0.15871338546276093, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04571465030312538, "sampling/sampling_logp_difference/max": 1.6292939186096191, "sampling/importance_sampling_ratio/min": 0.19606797397136688, "sampling/importance_sampling_ratio/mean": 0.9998616576194763, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23239293694496155, "clip_ratio/low_mean": 0.023657660058233887, "clip_ratio/low_min": 0.023657660058233887, "clip_ratio/high_mean": 0.018382353708148003, "clip_ratio/high_max": 0.018382353708148003, "clip_ratio/region_mean": 0.04204001376638189, "reward_total_mean": 0.7246866822242737, "reward_meter_mean": 0.9901050329208374, "reward_meter_std": 0.0037112515419721603, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.1608559489250183, "reward_total_composite_mean": 0.7246866822242737, "reward_total_composite_std": 0.15871340036392212} {"timestamp_utc": "2026-04-11T22:46:14Z", "mode": "train", "global_step": 716, "epoch": 0.02875848495802707, "loss": 0.0175, "grad_norm": 2.7540860176086426, "learning_rate": 7.833333333333333e-06, "num_tokens": 1573577.0, "completions/mean_length": 175.75, "completions/min_length": 162.0, "completions/max_length": 215.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 175.75, "completions/min_terminated_length": 162.0, "completions/max_terminated_length": 215.0, "rewards/meter/mean": 0.8235251903533936, "rewards/meter/std": 0.28830382227897644, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6091269850730896, "rewards/repeat_penalty/std": 0.19641855359077454, "rewards/total_composite/mean": 0.4638897478580475, "rewards/total_composite/std": 0.2275657206773758, "reward": 0.4638897478580475, "reward_std": 0.2275657057762146, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02091868966817856, "sampling/sampling_logp_difference/max": 2.359158992767334, "sampling/importance_sampling_ratio/min": 0.09449966251850128, "sampling/importance_sampling_ratio/mean": 1.0048706531524658, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15365072712302208, "clip_ratio/low_mean": 0.010836752247996628, "clip_ratio/low_min": 0.010836752247996628, "clip_ratio/high_mean": 0.0044064579415135086, "clip_ratio/high_max": 0.0044064579415135086, "clip_ratio/region_mean": 0.015243210189510137, "reward_total_mean": 0.4638897478580475, "reward_meter_mean": 0.8235251903533936, "reward_meter_std": 0.28830382227897644, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6091269850730896, "reward_repeat_penalty_std": 0.19641855359077454, "reward_total_composite_mean": 0.4638897478580475, "reward_total_composite_std": 0.2275657206773758} {"timestamp_utc": "2026-04-11T22:46:24Z", "mode": "train", "global_step": 717, "epoch": 0.028798650439812026, "loss": -0.1657, "grad_norm": 1.3285834789276123, "learning_rate": 7.83030303030303e-06, "num_tokens": 1575801.0, "completions/mean_length": 391.0, "completions/min_length": 156.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.625, "completions/mean_terminated_length": 189.33334350585938, "completions/min_terminated_length": 156.0, "completions/max_terminated_length": 210.0, "rewards/meter/mean": 0.3074490427970886, "rewards/meter/std": 0.2025083303451538, "rewards/count_adherence/mean": 0.824999988079071, "rewards/count_adherence/std": 0.12817399203777313, "rewards/arabic_clean/mean": 0.5, "rewards/arabic_clean/std": 0.5345224738121033, "rewards/repeat_penalty/mean": 0.8395833373069763, "rewards/repeat_penalty/std": 0.25571832060813904, "rewards/total_composite/mean": 0.08441510796546936, "rewards/total_composite/std": 0.10946666449308395, "reward": 0.08441510796546936, "reward_std": 0.10946667194366455, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05243195965886116, "sampling/sampling_logp_difference/max": 8.712602615356445, "sampling/importance_sampling_ratio/min": 0.0001644995791139081, "sampling/importance_sampling_ratio/mean": 1.002199649810791, "sampling/importance_sampling_ratio/max": 1.6081585884094238, "entropy": 0.13471947237849236, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.010442280676215887, "clip_ratio/high_max": 0.010442280676215887, "clip_ratio/region_mean": 0.010442280676215887, "reward_total_mean": 0.08441510796546936, "reward_meter_mean": 0.3074490427970886, "reward_meter_std": 0.2025083303451538, "reward_count_adherence_mean": 0.824999988079071, "reward_count_adherence_std": 0.12817399203777313, "reward_arabic_clean_mean": 0.5, "reward_arabic_clean_std": 0.5345224738121033, "reward_repeat_penalty_mean": 0.8395833373069763, "reward_repeat_penalty_std": 0.25571832060813904, "reward_total_composite_mean": 0.08441510796546936, "reward_total_composite_std": 0.10946666449308395} {"timestamp_utc": "2026-04-11T22:46:29Z", "mode": "train", "global_step": 718, "epoch": 0.02883881592159698, "loss": 0.0058, "grad_norm": 6.637593746185303, "learning_rate": 7.827272727272728e-06, "num_tokens": 1577519.0, "completions/mean_length": 57.75, "completions/min_length": 57.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.75, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9619162082672119, "rewards/meter/std": 0.06612562388181686, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9619162082672119, "rewards/total_composite/std": 0.06612562388181686, "reward": 0.9619162082672119, "reward_std": 0.06612562388181686, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03136257454752922, "sampling/sampling_logp_difference/max": 1.5830154418945312, "sampling/importance_sampling_ratio/min": 0.20535492897033691, "sampling/importance_sampling_ratio/mean": 1.004167914390564, "sampling/importance_sampling_ratio/max": 1.769487977027893, "entropy": 0.14472659677267075, "clip_ratio/low_mean": 0.006465517450124025, "clip_ratio/low_min": 0.006465517450124025, "clip_ratio/high_mean": 0.023710741428658366, "clip_ratio/high_max": 0.023710741428658366, "clip_ratio/region_mean": 0.03017625887878239, "reward_total_mean": 0.9619162082672119, "reward_meter_mean": 0.9619162082672119, "reward_meter_std": 0.06612562388181686, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9619162082672119, "reward_total_composite_std": 0.06612562388181686} {"timestamp_utc": "2026-04-11T22:46:39Z", "mode": "train", "global_step": 719, "epoch": 0.028878981403381934, "loss": -0.1972, "grad_norm": 0.982711136341095, "learning_rate": 7.824242424242425e-06, "num_tokens": 1580014.0, "completions/mean_length": 208.875, "completions/min_length": 142.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 165.57144165039062, "completions/min_terminated_length": 142.0, "completions/max_terminated_length": 171.0, "rewards/meter/mean": 0.8615807294845581, "rewards/meter/std": 0.1756223738193512, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.5022321939468384, "rewards/repeat_penalty/std": 0.19195686280727386, "rewards/total_composite/mean": 0.36472243070602417, "rewards/total_composite/std": 0.1904788464307785, "reward": 0.36472243070602417, "reward_std": 0.1904788315296173, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011406843550503254, "sampling/sampling_logp_difference/max": 0.5394062995910645, "sampling/importance_sampling_ratio/min": 0.5830943584442139, "sampling/importance_sampling_ratio/mean": 1.0052262544631958, "sampling/importance_sampling_ratio/max": 1.6336287260055542, "entropy": 0.07858177460730076, "clip_ratio/low_mean": 0.0008802816737443209, "clip_ratio/low_min": 0.0008802816737443209, "clip_ratio/high_mean": 0.005198180675506592, "clip_ratio/high_max": 0.005198180675506592, "clip_ratio/region_mean": 0.006078462349250913, "reward_total_mean": 0.36472243070602417, "reward_meter_mean": 0.8615807294845581, "reward_meter_std": 0.1756223738193512, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.5022321939468384, "reward_repeat_penalty_std": 0.19195686280727386, "reward_total_composite_mean": 0.36472243070602417, "reward_total_composite_std": 0.1904788464307785} {"timestamp_utc": "2026-04-11T22:46:47Z", "mode": "train", "global_step": 720, "epoch": 0.028919146885166887, "loss": -0.0478, "grad_norm": 1.5212147235870361, "learning_rate": 7.821212121212122e-06, "num_tokens": 1583419.0, "completions/mean_length": 228.625, "completions/min_length": 195.0, "completions/max_length": 243.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 228.625, "completions/min_terminated_length": 195.0, "completions/max_terminated_length": 243.0, "rewards/meter/mean": 0.9813181161880493, "rewards/meter/std": 0.009512215852737427, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.06613000482320786, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.527634859085083, "rewards/repeat_penalty/std": 0.18670302629470825, "rewards/total_composite/mean": 0.3818710744380951, "rewards/total_composite/std": 0.13756081461906433, "reward": 0.3818710744380951, "reward_std": 0.13756079971790314, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014808963052928448, "sampling/sampling_logp_difference/max": 1.9443225860595703, "sampling/importance_sampling_ratio/min": 0.14308412373065948, "sampling/importance_sampling_ratio/mean": 1.0003083944320679, "sampling/importance_sampling_ratio/max": 1.7837820053100586, "entropy": 0.08757321583107114, "clip_ratio/low_mean": 0.0012820513220503926, "clip_ratio/low_min": 0.0012820513220503926, "clip_ratio/high_mean": 0.015587894711643457, "clip_ratio/high_max": 0.015587894711643457, "clip_ratio/region_mean": 0.01686994603369385, "reward_total_mean": 0.3818710744380951, "reward_meter_mean": 0.9813181161880493, "reward_meter_std": 0.009512215852737427, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.06613000482320786, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.527634859085083, "reward_repeat_penalty_std": 0.18670302629470825, "reward_total_composite_mean": 0.3818710744380951, "reward_total_composite_std": 0.13756081461906433} {"timestamp_utc": "2026-04-11T22:46:52Z", "mode": "train", "global_step": 721, "epoch": 0.02895931236695184, "loss": -0.0114, "grad_norm": 9.25263786315918, "learning_rate": 7.81818181818182e-06, "num_tokens": 1585267.0, "completions/mean_length": 65.0, "completions/min_length": 60.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.0, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.6475868821144104, "rewards/meter/std": 0.3421926498413086, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.5946333408355713, "rewards/total_composite/std": 0.34638410806655884, "reward": 0.5946333408355713, "reward_std": 0.34638410806655884, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1203828752040863, "sampling/sampling_logp_difference/max": 2.941412925720215, "sampling/importance_sampling_ratio/min": 0.052791085094213486, "sampling/importance_sampling_ratio/mean": 0.991131603717804, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5916384495794773, "clip_ratio/low_mean": 0.051155281253159046, "clip_ratio/low_min": 0.051155281253159046, "clip_ratio/high_mean": 0.05381742771714926, "clip_ratio/high_max": 0.05381742771714926, "clip_ratio/region_mean": 0.1049727089703083, "reward_total_mean": 0.5946333408355713, "reward_meter_mean": 0.6475868821144104, "reward_meter_std": 0.3421926498413086, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_total_composite_mean": 0.5946333408355713, "reward_total_composite_std": 0.34638410806655884} {"timestamp_utc": "2026-04-11T22:46:57Z", "mode": "train", "global_step": 722, "epoch": 0.028999477848736795, "loss": 0.3273, "grad_norm": 7.886468410491943, "learning_rate": 7.815151515151515e-06, "num_tokens": 1586837.0, "completions/mean_length": 48.25, "completions/min_length": 42.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.25, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.9944906830787659, "rewards/meter/std": 0.0011895333882421255, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8698122501373291, "rewards/total_composite/std": 0.35145723819732666, "reward": 0.8698122501373291, "reward_std": 0.35145723819732666, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01547156646847725, "sampling/sampling_logp_difference/max": 0.7411696910858154, "sampling/importance_sampling_ratio/min": 0.4765561819076538, "sampling/importance_sampling_ratio/mean": 1.0014286041259766, "sampling/importance_sampling_ratio/max": 1.3288551568984985, "entropy": 0.10689277853816748, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8698122501373291, "reward_meter_mean": 0.9944906830787659, "reward_meter_std": 0.0011895333882421255, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8698122501373291, "reward_total_composite_std": 0.35145723819732666} {"timestamp_utc": "2026-04-11T22:47:08Z", "mode": "train", "global_step": 723, "epoch": 0.02903964333052175, "loss": -0.2014, "grad_norm": 0.8348656892776489, "learning_rate": 7.812121212121213e-06, "num_tokens": 1590899.0, "completions/mean_length": 485.75, "completions/min_length": 452.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 470.0, "completions/min_terminated_length": 452.0, "completions/max_terminated_length": 494.0, "rewards/meter/mean": 0.996362566947937, "rewards/meter/std": 0.0027744658291339874, "rewards/count_adherence/mean": 0.9464285373687744, "rewards/count_adherence/std": 0.03306501731276512, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.10470085591077805, "rewards/repeat_penalty/std": 0.10408195108175278, "rewards/total_composite/mean": 0.09887628257274628, "rewards/total_composite/std": 0.09698235988616943, "reward": 0.09887628257274628, "reward_std": 0.09698235988616943, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007093369495123625, "sampling/sampling_logp_difference/max": 1.990865707397461, "sampling/importance_sampling_ratio/min": 0.13657712936401367, "sampling/importance_sampling_ratio/mean": 1.0003700256347656, "sampling/importance_sampling_ratio/max": 1.536428451538086, "entropy": 0.018016068963333964, "clip_ratio/low_mean": 0.0007961043156683445, "clip_ratio/low_min": 0.0007961043156683445, "clip_ratio/high_mean": 0.0007826215587556362, "clip_ratio/high_max": 0.0007826215587556362, "clip_ratio/region_mean": 0.0015787258744239807, "reward_total_mean": 0.09887628257274628, "reward_meter_mean": 0.996362566947937, "reward_meter_std": 0.0027744658291339874, "reward_count_adherence_mean": 0.9464285373687744, "reward_count_adherence_std": 0.03306501731276512, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.10470085591077805, "reward_repeat_penalty_std": 0.10408195108175278, "reward_total_composite_mean": 0.09887628257274628, "reward_total_composite_std": 0.09698235988616943} {"timestamp_utc": "2026-04-11T22:47:17Z", "mode": "train", "global_step": 724, "epoch": 0.029079808812306703, "loss": 0.0117, "grad_norm": 1.2350369691848755, "learning_rate": 7.80909090909091e-06, "num_tokens": 1595768.0, "completions/mean_length": 390.625, "completions/min_length": 379.0, "completions/max_length": 403.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 390.625, "completions/min_terminated_length": 379.0, "completions/max_terminated_length": 403.0, "rewards/meter/mean": 0.9978216886520386, "rewards/meter/std": 0.0007184247369877994, "rewards/count_adherence/mean": 0.734375, "rewards/count_adherence/std": 0.04419417306780815, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.17863407731056213, "rewards/repeat_penalty/std": 0.16910506784915924, "rewards/total_composite/mean": 0.1255086362361908, "rewards/total_composite/std": 0.10828039050102234, "reward": 0.1255086362361908, "reward_std": 0.10828038305044174, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0057006035931408405, "sampling/sampling_logp_difference/max": 1.5134329795837402, "sampling/importance_sampling_ratio/min": 0.22015291452407837, "sampling/importance_sampling_ratio/mean": 0.9993754625320435, "sampling/importance_sampling_ratio/max": 1.8996793031692505, "entropy": 0.028950548847205937, "clip_ratio/low_mean": 0.0028556776233017445, "clip_ratio/low_min": 0.0028556776233017445, "clip_ratio/high_mean": 0.002920488826930523, "clip_ratio/high_max": 0.002920488826930523, "clip_ratio/region_mean": 0.005776166450232267, "reward_total_mean": 0.1255086362361908, "reward_meter_mean": 0.9978216886520386, "reward_meter_std": 0.0007184247369877994, "reward_count_adherence_mean": 0.734375, "reward_count_adherence_std": 0.04419417306780815, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.17863407731056213, "reward_repeat_penalty_std": 0.16910506784915924, "reward_total_composite_mean": 0.1255086362361908, "reward_total_composite_std": 0.10828039050102234} {"timestamp_utc": "2026-04-11T22:47:26Z", "mode": "train", "global_step": 725, "epoch": 0.029119974294091657, "loss": -0.1213, "grad_norm": 1.6505030393600464, "learning_rate": 7.806060606060607e-06, "num_tokens": 1597686.0, "completions/mean_length": 136.75, "completions/min_length": 78.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 83.14286041259766, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.5868990421295166, "rewards/meter/std": 0.38611555099487305, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.5530991554260254, "rewards/total_composite/std": 0.40852445363998413, "reward": 0.5530991554260254, "reward_std": 0.4085244834423065, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022672384977340698, "sampling/sampling_logp_difference/max": 0.911320686340332, "sampling/importance_sampling_ratio/min": 0.4019929766654968, "sampling/importance_sampling_ratio/mean": 1.0019892454147339, "sampling/importance_sampling_ratio/max": 1.7016674280166626, "entropy": 0.09931796230375767, "clip_ratio/low_mean": 0.004787406767718494, "clip_ratio/low_min": 0.004787406767718494, "clip_ratio/high_mean": 0.007217321544885635, "clip_ratio/high_max": 0.007217321544885635, "clip_ratio/region_mean": 0.01200472831260413, "reward_total_mean": 0.5530991554260254, "reward_meter_mean": 0.5868990421295166, "reward_meter_std": 0.38611555099487305, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.5530991554260254, "reward_total_composite_std": 0.40852445363998413} {"timestamp_utc": "2026-04-11T22:47:32Z", "mode": "train", "global_step": 726, "epoch": 0.02916013977587661, "loss": -0.0023, "grad_norm": 1.0429009199142456, "learning_rate": 7.803030303030303e-06, "num_tokens": 1600008.0, "completions/mean_length": 127.25, "completions/min_length": 126.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.25, "completions/min_terminated_length": 126.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9302235841751099, "rewards/meter/std": 0.015423719771206379, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5, "rewards/repeat_penalty/std": 0.1511857956647873, "rewards/total_composite/mean": 0.46501004695892334, "rewards/total_composite/std": 0.14159858226776123, "reward": 0.46501004695892334, "reward_std": 0.14159856736660004, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01078812312334776, "sampling/sampling_logp_difference/max": 0.9157018661499023, "sampling/importance_sampling_ratio/min": 0.4002356231212616, "sampling/importance_sampling_ratio/mean": 1.002754807472229, "sampling/importance_sampling_ratio/max": 1.5183614492416382, "entropy": 0.05762754753232002, "clip_ratio/low_mean": 0.0029605674790218472, "clip_ratio/low_min": 0.0029605674790218472, "clip_ratio/high_mean": 0.0009765625, "clip_ratio/high_max": 0.0009765625, "clip_ratio/region_mean": 0.003937129979021847, "reward_total_mean": 0.46501004695892334, "reward_meter_mean": 0.9302235841751099, "reward_meter_std": 0.015423719771206379, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5, "reward_repeat_penalty_std": 0.1511857956647873, "reward_total_composite_mean": 0.46501004695892334, "reward_total_composite_std": 0.14159858226776123} {"timestamp_utc": "2026-04-11T22:47:37Z", "mode": "train", "global_step": 727, "epoch": 0.029200305257661565, "loss": 0.0279, "grad_norm": 7.388904094696045, "learning_rate": 7.800000000000002e-06, "num_tokens": 1601923.0, "completions/mean_length": 99.375, "completions/min_length": 93.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.375, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.4179952144622803, "rewards/meter/std": 0.45823782682418823, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.3363860249519348, "rewards/total_composite/std": 0.36587125062942505, "reward": 0.3363860249519348, "reward_std": 0.36587125062942505, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0847281664609909, "sampling/sampling_logp_difference/max": 5.028669357299805, "sampling/importance_sampling_ratio/min": 0.006547517143189907, "sampling/importance_sampling_ratio/mean": 0.9850558042526245, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2684608269482851, "clip_ratio/low_mean": 0.032587712397798896, "clip_ratio/low_min": 0.032587712397798896, "clip_ratio/high_mean": 0.02533797360956669, "clip_ratio/high_max": 0.02533797360956669, "clip_ratio/region_mean": 0.057925686007365584, "reward_total_mean": 0.3363860249519348, "reward_meter_mean": 0.4179952144622803, "reward_meter_std": 0.45823782682418823, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.10690449178218842, "reward_total_composite_mean": 0.3363860249519348, "reward_total_composite_std": 0.36587125062942505} {"timestamp_utc": "2026-04-11T22:47:46Z", "mode": "train", "global_step": 728, "epoch": 0.02924047073944652, "loss": 0.0076, "grad_norm": 2.0805506706237793, "learning_rate": 7.796969696969697e-06, "num_tokens": 1607332.0, "completions/mean_length": 459.125, "completions/min_length": 427.0, "completions/max_length": 483.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 459.125, "completions/min_terminated_length": 427.0, "completions/max_terminated_length": 483.0, "rewards/meter/mean": 0.28689688444137573, "rewards/meter/std": 0.3819184899330139, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.0431290864944458, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5729086995124817, "rewards/repeat_penalty/std": 0.018385794013738632, "rewards/total_composite/mean": 0.12284211814403534, "rewards/total_composite/std": 0.1601785272359848, "reward": 0.12284211814403534, "reward_std": 0.1601785272359848, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019973881542682648, "sampling/sampling_logp_difference/max": 3.37805438041687, "sampling/importance_sampling_ratio/min": 0.03411376476287842, "sampling/importance_sampling_ratio/mean": 0.9993253350257874, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06644301256164908, "clip_ratio/low_mean": 0.008432107453700155, "clip_ratio/low_min": 0.008432107453700155, "clip_ratio/high_mean": 0.003258531214669347, "clip_ratio/high_max": 0.003258531214669347, "clip_ratio/region_mean": 0.011690638668369502, "reward_total_mean": 0.12284211814403534, "reward_meter_mean": 0.28689688444137573, "reward_meter_std": 0.3819184899330139, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.0431290864944458, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5729086995124817, "reward_repeat_penalty_std": 0.018385794013738632, "reward_total_composite_mean": 0.12284211814403534, "reward_total_composite_std": 0.1601785272359848} {"timestamp_utc": "2026-04-11T22:47:56Z", "mode": "train", "global_step": 729, "epoch": 0.029280636221231473, "loss": -0.1369, "grad_norm": 2.659188747406006, "learning_rate": 7.793939393939394e-06, "num_tokens": 1609182.0, "completions/mean_length": 139.25, "completions/min_length": 80.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 86.00000762939453, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.700344443321228, "rewards/meter/std": 0.42837557196617126, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6998236179351807, "rewards/total_composite/std": 0.4293443560600281, "reward": 0.6998236179351807, "reward_std": 0.4293443262577057, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022995855659246445, "sampling/sampling_logp_difference/max": 0.6324782371520996, "sampling/importance_sampling_ratio/min": 0.5312735438346863, "sampling/importance_sampling_ratio/mean": 1.0047773122787476, "sampling/importance_sampling_ratio/max": 1.5019832849502563, "entropy": 0.1982464650645852, "clip_ratio/low_mean": 0.0015625000232830644, "clip_ratio/low_min": 0.0015625000232830644, "clip_ratio/high_mean": 0.007218867307528853, "clip_ratio/high_max": 0.007218867307528853, "clip_ratio/region_mean": 0.008781367330811918, "reward_total_mean": 0.6998236179351807, "reward_meter_mean": 0.700344443321228, "reward_meter_std": 0.42837557196617126, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6998236179351807, "reward_total_composite_std": 0.4293443560600281} {"timestamp_utc": "2026-04-11T22:48:01Z", "mode": "train", "global_step": 730, "epoch": 0.029320801703016427, "loss": 0.0222, "grad_norm": 3.446493625640869, "learning_rate": 7.790909090909092e-06, "num_tokens": 1611175.0, "completions/mean_length": 84.125, "completions/min_length": 78.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.125, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.9330159425735474, "rewards/meter/std": 0.04698435589671135, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9330159425735474, "rewards/total_composite/std": 0.04698435589671135, "reward": 0.9330159425735474, "reward_std": 0.04698435962200165, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02119125984609127, "sampling/sampling_logp_difference/max": 1.4188241958618164, "sampling/importance_sampling_ratio/min": 0.3314521312713623, "sampling/importance_sampling_ratio/mean": 1.0041048526763916, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.13521220535039902, "clip_ratio/low_mean": 0.002890269970521331, "clip_ratio/low_min": 0.002890269970521331, "clip_ratio/high_mean": 0.010773762594908476, "clip_ratio/high_max": 0.010773762594908476, "clip_ratio/region_mean": 0.013664032565429807, "reward_total_mean": 0.9330159425735474, "reward_meter_mean": 0.9330159425735474, "reward_meter_std": 0.04698435589671135, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9330159425735474, "reward_total_composite_std": 0.04698435589671135} {"timestamp_utc": "2026-04-11T22:48:11Z", "mode": "train", "global_step": 731, "epoch": 0.02936096718480138, "loss": -0.2761, "grad_norm": 3.5135695934295654, "learning_rate": 7.787878787878789e-06, "num_tokens": 1613214.0, "completions/mean_length": 506.875, "completions/min_length": 471.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.875, "completions/mean_terminated_length": 471.0, "completions/min_terminated_length": 471.0, "completions/max_terminated_length": 471.0, "rewards/meter/mean": 0.9958064556121826, "rewards/meter/std": 0.0007912082364782691, "rewards/count_adherence/mean": 0.5277777910232544, "rewards/count_adherence/std": 0.051434461027383804, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.11126373708248138, "rewards/repeat_penalty/std": 0.07416856288909912, "rewards/total_composite/mean": 0.055620092898607254, "rewards/total_composite/std": 0.029501251876354218, "reward": 0.055620092898607254, "reward_std": 0.02950124815106392, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013655466958880424, "sampling/sampling_logp_difference/max": 1.3458614349365234, "sampling/importance_sampling_ratio/min": 0.2603153586387634, "sampling/importance_sampling_ratio/mean": 0.9985577464103699, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.005113576073199511, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0005307855899445713, "clip_ratio/high_max": 0.0005307855899445713, "clip_ratio/region_mean": 0.0005307855899445713, "reward_total_mean": 0.055620092898607254, "reward_meter_mean": 0.9958064556121826, "reward_meter_std": 0.0007912082364782691, "reward_count_adherence_mean": 0.5277777910232544, "reward_count_adherence_std": 0.051434461027383804, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.11126373708248138, "reward_repeat_penalty_std": 0.07416856288909912, "reward_total_composite_mean": 0.055620092898607254, "reward_total_composite_std": 0.029501251876354218} {"timestamp_utc": "2026-04-11T22:48:21Z", "mode": "train", "global_step": 732, "epoch": 0.029401132666586335, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.784848484848484e-06, "num_tokens": 1614982.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9962900280952454, "rewards/meter/std": 0.001235451316460967, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0345032773911953, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.19743433594703674, "rewards/repeat_penalty/std": 0.17679214477539062, "rewards/total_composite/mean": 0.18774111568927765, "rewards/total_composite/std": 0.16288606822490692, "reward": 0.18774111568927765, "reward_std": 0.16288605332374573, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.18774111568927765, "reward_meter_mean": 0.9962900280952454, "reward_meter_std": 0.001235451316460967, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0345032773911953, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.19743433594703674, "reward_repeat_penalty_std": 0.17679214477539062, "reward_total_composite_mean": 0.18774111568927765, "reward_total_composite_std": 0.16288606822490692} {"timestamp_utc": "2026-04-11T22:48:26Z", "mode": "train", "global_step": 733, "epoch": 0.02944129814837129, "loss": 0.026, "grad_norm": 5.411409378051758, "learning_rate": 7.781818181818183e-06, "num_tokens": 1616806.0, "completions/mean_length": 72.0, "completions/min_length": 63.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.0, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.6948316097259521, "rewards/meter/std": 0.2794814109802246, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8333333730697632, "rewards/repeat_penalty/std": 0.17817415297031403, "rewards/total_composite/mean": 0.5835638046264648, "rewards/total_composite/std": 0.28273841738700867, "reward": 0.5835638046264648, "reward_std": 0.28273844718933105, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.055757418274879456, "sampling/sampling_logp_difference/max": 3.232696771621704, "sampling/importance_sampling_ratio/min": 0.03945096582174301, "sampling/importance_sampling_ratio/mean": 1.0083332061767578, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3269932046532631, "clip_ratio/low_mean": 0.034895967692136765, "clip_ratio/low_min": 0.034895967692136765, "clip_ratio/high_mean": 0.027756539173424244, "clip_ratio/high_max": 0.027756539173424244, "clip_ratio/region_mean": 0.06265250686556101, "reward_total_mean": 0.5835638046264648, "reward_meter_mean": 0.6948316097259521, "reward_meter_std": 0.2794814109802246, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8333333730697632, "reward_repeat_penalty_std": 0.17817415297031403, "reward_total_composite_mean": 0.5835638046264648, "reward_total_composite_std": 0.28273841738700867} {"timestamp_utc": "2026-04-11T22:48:31Z", "mode": "train", "global_step": 734, "epoch": 0.029481463630156243, "loss": 0.0012, "grad_norm": 1.1426414251327515, "learning_rate": 7.778787878787879e-06, "num_tokens": 1619174.0, "completions/mean_length": 116.0, "completions/min_length": 116.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.0, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.9807324409484863, "rewards/meter/std": 0.005540105979889631, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8406277894973755, "rewards/total_composite/std": 0.004748670384287834, "reward": 0.8406277894973755, "reward_std": 0.004748655948787928, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004732904955744743, "sampling/sampling_logp_difference/max": 0.8726806640625, "sampling/importance_sampling_ratio/min": 0.4178299903869629, "sampling/importance_sampling_ratio/mean": 1.0012872219085693, "sampling/importance_sampling_ratio/max": 1.4202854633331299, "entropy": 0.025226276833564043, "clip_ratio/low_mean": 0.0021551724057644606, "clip_ratio/low_min": 0.0021551724057644606, "clip_ratio/high_mean": 0.0010775862028822303, "clip_ratio/high_max": 0.0010775862028822303, "clip_ratio/region_mean": 0.003232758608646691, "reward_total_mean": 0.8406277894973755, "reward_meter_mean": 0.9807324409484863, "reward_meter_std": 0.005540105979889631, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8406277894973755, "reward_total_composite_std": 0.004748670384287834} {"timestamp_utc": "2026-04-11T22:48:36Z", "mode": "train", "global_step": 735, "epoch": 0.029521629111941197, "loss": 0.0021, "grad_norm": 6.164267539978027, "learning_rate": 7.775757575757576e-06, "num_tokens": 1620938.0, "completions/mean_length": 77.5, "completions/min_length": 74.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.5, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.8432132005691528, "rewards/meter/std": 0.11800947040319443, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9249999523162842, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.7734056115150452, "rewards/total_composite/std": 0.09497449547052383, "reward": 0.7734056115150452, "reward_std": 0.09497448056936264, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04740333557128906, "sampling/sampling_logp_difference/max": 1.3389692306518555, "sampling/importance_sampling_ratio/min": 0.2621157169342041, "sampling/importance_sampling_ratio/mean": 1.0057860612869263, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2497086301445961, "clip_ratio/low_mean": 0.026082158088684082, "clip_ratio/low_min": 0.026082158088684082, "clip_ratio/high_mean": 0.011470985249616206, "clip_ratio/high_max": 0.011470985249616206, "clip_ratio/region_mean": 0.03755314333830029, "reward_total_mean": 0.7734056115150452, "reward_meter_mean": 0.8432132005691528, "reward_meter_std": 0.11800947040319443, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9249999523162842, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.7734056115150452, "reward_total_composite_std": 0.09497449547052383} {"timestamp_utc": "2026-04-11T22:48:44Z", "mode": "train", "global_step": 736, "epoch": 0.02956179459372615, "loss": 0.46, "grad_norm": 5.787602424621582, "learning_rate": 7.772727272727273e-06, "num_tokens": 1623113.0, "completions/mean_length": 115.875, "completions/min_length": 72.0, "completions/max_length": 353.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.875, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 353.0, "rewards/meter/mean": 0.7237052917480469, "rewards/meter/std": 0.3117098808288574, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6369646191596985, "rewards/total_composite/std": 0.40405309200286865, "reward": 0.6369646191596985, "reward_std": 0.40405309200286865, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08920754492282867, "sampling/sampling_logp_difference/max": 1.3201618194580078, "sampling/importance_sampling_ratio/min": 0.2670920789241791, "sampling/importance_sampling_ratio/mean": 1.0180284976959229, "sampling/importance_sampling_ratio/max": 1.964148998260498, "entropy": 1.2863622568547726, "clip_ratio/low_mean": 0.008429729146882892, "clip_ratio/low_min": 0.008429729146882892, "clip_ratio/high_mean": 0.01677176496013999, "clip_ratio/high_max": 0.01677176496013999, "clip_ratio/region_mean": 0.02520149410702288, "reward_total_mean": 0.6369646191596985, "reward_meter_mean": 0.7237052917480469, "reward_meter_std": 0.3117098808288574, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6369646191596985, "reward_total_composite_std": 0.40405309200286865} {"timestamp_utc": "2026-04-11T22:48:54Z", "mode": "train", "global_step": 737, "epoch": 0.029601960075511104, "loss": -0.2132, "grad_norm": 1.1047923564910889, "learning_rate": 7.76969696969697e-06, "num_tokens": 1625144.0, "completions/mean_length": 294.875, "completions/min_length": 147.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 164.60000610351562, "completions/min_terminated_length": 147.0, "completions/max_terminated_length": 191.0, "rewards/meter/mean": 0.3803099989891052, "rewards/meter/std": 0.2973701059818268, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.22160132229328156, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/repeat_penalty/mean": 0.797619104385376, "rewards/repeat_penalty/std": 0.2185886949300766, "rewards/total_composite/mean": 0.20652014017105103, "rewards/total_composite/std": 0.19584397971630096, "reward": 0.20652014017105103, "reward_std": 0.19584397971630096, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030263110995292664, "sampling/sampling_logp_difference/max": 1.4690535068511963, "sampling/importance_sampling_ratio/min": 0.2301432192325592, "sampling/importance_sampling_ratio/mean": 1.0015567541122437, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12422919739037752, "clip_ratio/low_mean": 0.0008333333535119891, "clip_ratio/low_min": 0.0008333333535119891, "clip_ratio/high_mean": 0.015350477071478963, "clip_ratio/high_max": 0.015350477071478963, "clip_ratio/region_mean": 0.016183810424990952, "reward_total_mean": 0.20652014017105103, "reward_meter_mean": 0.3803099989891052, "reward_meter_std": 0.2973701059818268, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.22160132229328156, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_repeat_penalty_mean": 0.797619104385376, "reward_repeat_penalty_std": 0.2185886949300766, "reward_total_composite_mean": 0.20652014017105103, "reward_total_composite_std": 0.19584397971630096} {"timestamp_utc": "2026-04-11T22:49:04Z", "mode": "train", "global_step": 738, "epoch": 0.02964212555729606, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.766666666666666e-06, "num_tokens": 1626768.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9730854630470276, "rewards/meter/std": 0.03588658943772316, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.050507619976997375, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.2781907618045807, "rewards/repeat_penalty/std": 0.2387264519929886, "rewards/total_composite/mean": 0.2381971776485443, "rewards/total_composite/std": 0.20977550745010376, "reward": 0.2381971776485443, "reward_std": 0.20977550745010376, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.2381971776485443, "reward_meter_mean": 0.9730854630470276, "reward_meter_std": 0.03588658943772316, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.050507619976997375, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.2781907618045807, "reward_repeat_penalty_std": 0.2387264519929886, "reward_total_composite_mean": 0.2381971776485443, "reward_total_composite_std": 0.20977550745010376} {"timestamp_utc": "2026-04-11T22:49:13Z", "mode": "train", "global_step": 739, "epoch": 0.029682291039081012, "loss": 0.0363, "grad_norm": 1.4499318599700928, "learning_rate": 7.763636363636364e-06, "num_tokens": 1631595.0, "completions/mean_length": 423.375, "completions/min_length": 398.0, "completions/max_length": 483.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 423.375, "completions/min_terminated_length": 398.0, "completions/max_terminated_length": 483.0, "rewards/meter/mean": 0.7534134387969971, "rewards/meter/std": 0.2928631901741028, "rewards/count_adherence/mean": 0.3888888955116272, "rewards/count_adherence/std": 0.059391383081674576, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5623973608016968, "rewards/repeat_penalty/std": 0.18699996173381805, "rewards/total_composite/mean": 0.15555939078330994, "rewards/total_composite/std": 0.07782954722642899, "reward": 0.15555939078330994, "reward_std": 0.07782954722642899, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013467294164001942, "sampling/sampling_logp_difference/max": 3.2660930156707764, "sampling/importance_sampling_ratio/min": 0.0381552055478096, "sampling/importance_sampling_ratio/mean": 0.9997313022613525, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0637149391695857, "clip_ratio/low_mean": 0.005539899080758914, "clip_ratio/low_min": 0.005539899080758914, "clip_ratio/high_mean": 0.004910050658509135, "clip_ratio/high_max": 0.004910050658509135, "clip_ratio/region_mean": 0.01044994973926805, "reward_total_mean": 0.15555939078330994, "reward_meter_mean": 0.7534134387969971, "reward_meter_std": 0.2928631901741028, "reward_count_adherence_mean": 0.3888888955116272, "reward_count_adherence_std": 0.059391383081674576, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5623973608016968, "reward_repeat_penalty_std": 0.18699996173381805, "reward_total_composite_mean": 0.15555939078330994, "reward_total_composite_std": 0.07782954722642899} {"timestamp_utc": "2026-04-11T22:49:19Z", "mode": "train", "global_step": 740, "epoch": 0.029722456520865966, "loss": -0.0085, "grad_norm": 3.1338510513305664, "learning_rate": 7.76060606060606e-06, "num_tokens": 1634096.0, "completions/mean_length": 133.625, "completions/min_length": 126.0, "completions/max_length": 142.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 133.625, "completions/min_terminated_length": 126.0, "completions/max_terminated_length": 142.0, "rewards/meter/mean": 0.11201652884483337, "rewards/meter/std": 0.19264455139636993, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6964285969734192, "rewards/repeat_penalty/std": 0.11921756714582443, "rewards/total_composite/mean": 0.0800618976354599, "rewards/total_composite/std": 0.13757191598415375, "reward": 0.0800618976354599, "reward_std": 0.13757191598415375, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02300534024834633, "sampling/sampling_logp_difference/max": 1.0942306518554688, "sampling/importance_sampling_ratio/min": 0.33479708433151245, "sampling/importance_sampling_ratio/mean": 1.005319595336914, "sampling/importance_sampling_ratio/max": 1.8425383567810059, "entropy": 0.15879025869071484, "clip_ratio/low_mean": 0.008456611772999167, "clip_ratio/low_min": 0.008456611772999167, "clip_ratio/high_mean": 0.009530608775094151, "clip_ratio/high_max": 0.009530608775094151, "clip_ratio/region_mean": 0.01798722054809332, "reward_total_mean": 0.0800618976354599, "reward_meter_mean": 0.11201652884483337, "reward_meter_std": 0.19264455139636993, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6964285969734192, "reward_repeat_penalty_std": 0.11921756714582443, "reward_total_composite_mean": 0.0800618976354599, "reward_total_composite_std": 0.13757191598415375} {"timestamp_utc": "2026-04-11T22:49:24Z", "mode": "train", "global_step": 741, "epoch": 0.02976262200265092, "loss": -0.0061, "grad_norm": 1.2720859050750732, "learning_rate": 7.757575757575758e-06, "num_tokens": 1636098.0, "completions/mean_length": 79.25, "completions/min_length": 78.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.25, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9925205707550049, "rewards/meter/std": 0.0022588411811739206, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.794016420841217, "rewards/total_composite/std": 0.001807067426852882, "reward": 0.794016420841217, "reward_std": 0.0018070697551593184, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011411315761506557, "sampling/sampling_logp_difference/max": 0.8109602928161621, "sampling/importance_sampling_ratio/min": 0.4444310963153839, "sampling/importance_sampling_ratio/mean": 1.001147985458374, "sampling/importance_sampling_ratio/max": 1.5393544435501099, "entropy": 0.07466331776231527, "clip_ratio/low_mean": 0.006329114083200693, "clip_ratio/low_min": 0.006329114083200693, "clip_ratio/high_mean": 0.007794186705723405, "clip_ratio/high_max": 0.007794186705723405, "clip_ratio/region_mean": 0.014123300788924098, "reward_total_mean": 0.794016420841217, "reward_meter_mean": 0.9925205707550049, "reward_meter_std": 0.0022588411811739206, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.794016420841217, "reward_total_composite_std": 0.001807067426852882} {"timestamp_utc": "2026-04-11T22:49:29Z", "mode": "train", "global_step": 742, "epoch": 0.029802787484435874, "loss": 0.012, "grad_norm": 7.301284313201904, "learning_rate": 7.754545454545455e-06, "num_tokens": 1638424.0, "completions/mean_length": 120.75, "completions/min_length": 114.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.75, "completions/min_terminated_length": 114.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.7745586633682251, "rewards/meter/std": 0.25043216347694397, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6964285969734192, "rewards/repeat_penalty/std": 0.17806050181388855, "rewards/total_composite/mean": 0.5217785835266113, "rewards/total_composite/std": 0.1874629557132721, "reward": 0.5217785835266113, "reward_std": 0.1874629259109497, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04599983990192413, "sampling/sampling_logp_difference/max": 4.404999732971191, "sampling/importance_sampling_ratio/min": 0.012216109782457352, "sampling/importance_sampling_ratio/mean": 0.9962561726570129, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12627906422130764, "clip_ratio/low_mean": 0.021951976465061307, "clip_ratio/low_min": 0.021951976465061307, "clip_ratio/high_mean": 0.018717249389737844, "clip_ratio/high_max": 0.018717249389737844, "clip_ratio/region_mean": 0.04066922585479915, "reward_total_mean": 0.5217785835266113, "reward_meter_mean": 0.7745586633682251, "reward_meter_std": 0.25043216347694397, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6964285969734192, "reward_repeat_penalty_std": 0.17806050181388855, "reward_total_composite_mean": 0.5217785835266113, "reward_total_composite_std": 0.1874629557132721} {"timestamp_utc": "2026-04-11T22:49:39Z", "mode": "train", "global_step": 743, "epoch": 0.02984295296622083, "loss": 0.1769, "grad_norm": 1.0547409057617188, "learning_rate": 7.751515151515153e-06, "num_tokens": 1642208.0, "completions/mean_length": 510.0, "completions/min_length": 507.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.5, "completions/mean_terminated_length": 508.0, "completions/min_terminated_length": 507.0, "completions/max_terminated_length": 510.0, "rewards/meter/mean": 0.45572322607040405, "rewards/meter/std": 0.1519794911146164, "rewards/count_adherence/mean": 0.6153846383094788, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.5480158925056458, "rewards/repeat_penalty/std": 0.010451768524944782, "rewards/total_composite/mean": 0.13946115970611572, "rewards/total_composite/std": 0.07740650326013565, "reward": 0.13946115970611572, "reward_std": 0.07740650326013565, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00907099712640047, "sampling/sampling_logp_difference/max": 1.135289192199707, "sampling/importance_sampling_ratio/min": 0.32132917642593384, "sampling/importance_sampling_ratio/mean": 1.0025880336761475, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03308352828025818, "clip_ratio/low_mean": 0.0017248676158487797, "clip_ratio/low_min": 0.0017248676158487797, "clip_ratio/high_mean": 0.0012283907853998244, "clip_ratio/high_max": 0.0012283907853998244, "clip_ratio/region_mean": 0.002953258401248604, "reward_total_mean": 0.13946115970611572, "reward_meter_mean": 0.45572322607040405, "reward_meter_std": 0.1519794911146164, "reward_count_adherence_mean": 0.6153846383094788, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.5480158925056458, "reward_repeat_penalty_std": 0.010451768524944782, "reward_total_composite_mean": 0.13946115970611572, "reward_total_composite_std": 0.07740650326013565} {"timestamp_utc": "2026-04-11T22:49:44Z", "mode": "train", "global_step": 744, "epoch": 0.029883118448005785, "loss": 0.0283, "grad_norm": 6.8046875, "learning_rate": 7.74848484848485e-06, "num_tokens": 1643996.0, "completions/mean_length": 64.5, "completions/min_length": 61.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.5, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.5396110415458679, "rewards/meter/std": 0.20649512112140656, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5396110415458679, "rewards/total_composite/std": 0.20649512112140656, "reward": 0.5396110415458679, "reward_std": 0.20649512112140656, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06342907249927521, "sampling/sampling_logp_difference/max": 1.2414631843566895, "sampling/importance_sampling_ratio/min": 0.28896111249923706, "sampling/importance_sampling_ratio/mean": 1.0097684860229492, "sampling/importance_sampling_ratio/max": 1.632293462753296, "entropy": 0.6070369817316532, "clip_ratio/low_mean": 0.02111415727995336, "clip_ratio/low_min": 0.02111415727995336, "clip_ratio/high_mean": 0.027501578675583005, "clip_ratio/high_max": 0.027501578675583005, "clip_ratio/region_mean": 0.048615735955536366, "reward_total_mean": 0.5396110415458679, "reward_meter_mean": 0.5396110415458679, "reward_meter_std": 0.20649512112140656, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5396110415458679, "reward_total_composite_std": 0.20649512112140656} {"timestamp_utc": "2026-04-11T22:49:49Z", "mode": "train", "global_step": 745, "epoch": 0.02992328392979074, "loss": 0.0099, "grad_norm": 9.705404281616211, "learning_rate": 7.745454545454545e-06, "num_tokens": 1645756.0, "completions/mean_length": 63.0, "completions/min_length": 62.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9103853106498718, "rewards/meter/std": 0.22165298461914062, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9103853106498718, "rewards/total_composite/std": 0.22165298461914062, "reward": 0.9103853106498718, "reward_std": 0.22165298461914062, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07189808040857315, "sampling/sampling_logp_difference/max": 2.013796806335449, "sampling/importance_sampling_ratio/min": 0.13348090648651123, "sampling/importance_sampling_ratio/mean": 1.0033738613128662, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.25149067025631666, "clip_ratio/low_mean": 0.008064515888690948, "clip_ratio/low_min": 0.008064515888690948, "clip_ratio/high_mean": 0.043645198456943035, "clip_ratio/high_max": 0.043645198456943035, "clip_ratio/region_mean": 0.051709714345633984, "reward_total_mean": 0.9103853106498718, "reward_meter_mean": 0.9103853106498718, "reward_meter_std": 0.22165298461914062, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9103853106498718, "reward_total_composite_std": 0.22165298461914062} {"timestamp_utc": "2026-04-11T22:49:56Z", "mode": "train", "global_step": 746, "epoch": 0.029963449411575693, "loss": 0.0227, "grad_norm": 2.2934165000915527, "learning_rate": 7.742424242424244e-06, "num_tokens": 1649182.0, "completions/mean_length": 238.25, "completions/min_length": 205.0, "completions/max_length": 260.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 238.25, "completions/min_terminated_length": 205.0, "completions/max_terminated_length": 260.0, "rewards/meter/mean": 0.8676279783248901, "rewards/meter/std": 0.3215867578983307, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.36800700426101685, "rewards/repeat_penalty/std": 0.24577626585960388, "rewards/total_composite/mean": 0.2228499948978424, "rewards/total_composite/std": 0.18318407237529755, "reward": 0.2228499948978424, "reward_std": 0.18318407237529755, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018860070034861565, "sampling/sampling_logp_difference/max": 4.0910139083862305, "sampling/importance_sampling_ratio/min": 0.016722269356250763, "sampling/importance_sampling_ratio/mean": 1.0014268159866333, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06980151077732444, "clip_ratio/low_mean": 0.003769519622437656, "clip_ratio/low_min": 0.003769519622437656, "clip_ratio/high_mean": 0.006430621142499149, "clip_ratio/high_max": 0.006430621142499149, "clip_ratio/region_mean": 0.010200140764936805, "reward_total_mean": 0.2228499948978424, "reward_meter_mean": 0.8676279783248901, "reward_meter_std": 0.3215867578983307, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.36800700426101685, "reward_repeat_penalty_std": 0.24577626585960388, "reward_total_composite_mean": 0.2228499948978424, "reward_total_composite_std": 0.18318407237529755} {"timestamp_utc": "2026-04-11T22:50:00Z", "mode": "train", "global_step": 747, "epoch": 0.030003614893360647, "loss": -0.0252, "grad_norm": 7.876811981201172, "learning_rate": 7.73939393939394e-06, "num_tokens": 1650683.0, "completions/mean_length": 40.625, "completions/min_length": 38.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.625, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9663740396499634, "rewards/meter/std": 0.011263493448495865, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9663740396499634, "rewards/total_composite/std": 0.011263493448495865, "reward": 0.9663740396499634, "reward_std": 0.011263499036431313, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03788968175649643, "sampling/sampling_logp_difference/max": 0.8854336738586426, "sampling/importance_sampling_ratio/min": 0.41253525018692017, "sampling/importance_sampling_ratio/mean": 1.0029720067977905, "sampling/importance_sampling_ratio/max": 1.344473123550415, "entropy": 0.28702736645936966, "clip_ratio/low_mean": 0.012351190904155374, "clip_ratio/low_min": 0.012351190904155374, "clip_ratio/high_mean": 0.018227352295070887, "clip_ratio/high_max": 0.018227352295070887, "clip_ratio/region_mean": 0.03057854319922626, "reward_total_mean": 0.9663740396499634, "reward_meter_mean": 0.9663740396499634, "reward_meter_std": 0.011263493448495865, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9663740396499634, "reward_total_composite_std": 0.011263493448495865} {"timestamp_utc": "2026-04-11T22:50:05Z", "mode": "train", "global_step": 748, "epoch": 0.0300437803751456, "loss": 0.031, "grad_norm": 4.674757957458496, "learning_rate": 7.736363636363637e-06, "num_tokens": 1652564.0, "completions/mean_length": 82.125, "completions/min_length": 74.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.125, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.8491703271865845, "rewards/meter/std": 0.15907908976078033, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8491703271865845, "rewards/total_composite/std": 0.15907908976078033, "reward": 0.8491703271865845, "reward_std": 0.15907907485961914, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0295439250767231, "sampling/sampling_logp_difference/max": 1.300436019897461, "sampling/importance_sampling_ratio/min": 0.2724129855632782, "sampling/importance_sampling_ratio/mean": 1.0036686658859253, "sampling/importance_sampling_ratio/max": 1.843922734260559, "entropy": 0.22462552785873413, "clip_ratio/low_mean": 0.002961171790957451, "clip_ratio/low_min": 0.002961171790957451, "clip_ratio/high_mean": 0.021474606008268893, "clip_ratio/high_max": 0.021474606008268893, "clip_ratio/region_mean": 0.024435777799226344, "reward_total_mean": 0.8491703271865845, "reward_meter_mean": 0.8491703271865845, "reward_meter_std": 0.15907908976078033, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8491703271865845, "reward_total_composite_std": 0.15907908976078033} {"timestamp_utc": "2026-04-11T22:50:10Z", "mode": "train", "global_step": 749, "epoch": 0.030083945856930555, "loss": 0.0372, "grad_norm": 6.799373626708984, "learning_rate": 7.733333333333334e-06, "num_tokens": 1654625.0, "completions/mean_length": 93.625, "completions/min_length": 87.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.625, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9564855694770813, "rewards/meter/std": 0.10235674679279327, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.8075714707374573, "rewards/total_composite/std": 0.0810769721865654, "reward": 0.8075714707374573, "reward_std": 0.0810769572854042, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03427453711628914, "sampling/sampling_logp_difference/max": 2.4858551025390625, "sampling/importance_sampling_ratio/min": 0.08325432986021042, "sampling/importance_sampling_ratio/mean": 0.9965777397155762, "sampling/importance_sampling_ratio/max": 1.903843641281128, "entropy": 0.12411046819761395, "clip_ratio/low_mean": 0.026110241888090968, "clip_ratio/low_min": 0.026110241888090968, "clip_ratio/high_mean": 0.008620689623057842, "clip_ratio/high_max": 0.008620689623057842, "clip_ratio/region_mean": 0.03473093151114881, "reward_total_mean": 0.8075714707374573, "reward_meter_mean": 0.9564855694770813, "reward_meter_std": 0.10235674679279327, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.8075714707374573, "reward_total_composite_std": 0.0810769721865654} {"timestamp_utc": "2026-04-11T22:50:15Z", "mode": "train", "global_step": 750, "epoch": 0.03012411133871551, "loss": 0.031, "grad_norm": 6.688387870788574, "learning_rate": 7.730303030303032e-06, "num_tokens": 1656513.0, "completions/mean_length": 67.0, "completions/min_length": 65.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9406866431236267, "rewards/meter/std": 0.1574798822402954, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.8990964889526367, "rewards/total_composite/std": 0.18213681876659393, "reward": 0.8990964889526367, "reward_std": 0.18213684856891632, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.035663776099681854, "sampling/sampling_logp_difference/max": 0.9646925926208496, "sampling/importance_sampling_ratio/min": 0.42169255018234253, "sampling/importance_sampling_ratio/mean": 1.0121128559112549, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.21013184823095798, "clip_ratio/low_mean": 0.00892857147846371, "clip_ratio/low_min": 0.00892857147846371, "clip_ratio/high_mean": 0.022817256744019687, "clip_ratio/high_max": 0.022817256744019687, "clip_ratio/region_mean": 0.031745828222483397, "reward_total_mean": 0.8990964889526367, "reward_meter_mean": 0.9406866431236267, "reward_meter_std": 0.1574798822402954, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.8990964889526367, "reward_total_composite_std": 0.18213681876659393} {"timestamp_utc": "2026-04-11T22:51:48Z", "mode": "eval", "global_step": 750, "epoch": 0.03012411133871551, "eval_loss": NaN, "eval_runtime": 93.1633, "eval_samples_per_second": 1.116, "eval_steps_per_second": 0.14, "eval_num_tokens": 1656513.0, "eval_completions/mean_length": 284.52884615384613, "eval_completions/min_length": 61.46153846153846, "eval_completions/max_length": 496.7692307692308, "eval_completions/clipped_ratio": 0.23076923076923078, "eval_completions/mean_terminated_length": 215.78288092980017, "eval_completions/min_terminated_length": 61.46153846153846, "eval_completions/max_terminated_length": 393.46153846153845, "eval_rewards/meter/mean": 0.6660121427132533, "eval_rewards/meter/std": 0.37975076070198643, "eval_rewards/count_adherence/mean": 0.7424245018225449, "eval_rewards/count_adherence/std": 0.2582179422561939, "eval_rewards/arabic_clean/mean": 0.9711538461538461, "eval_rewards/arabic_clean/std": 0.08158924258672275, "eval_rewards/repeat_penalty/mean": 0.638540084545429, "eval_rewards/repeat_penalty/std": 0.270676647241299, "eval_rewards/total_composite/mean": 0.33899185634576356, "eval_rewards/total_composite/std": 0.3308297275350644, "eval_reward": 0.33899185634576356, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.012646582407447008, "eval_sampling/sampling_logp_difference/max": 0.8113565261547382, "eval_sampling/importance_sampling_ratio/min": 0.4615517258644104, "eval_sampling/importance_sampling_ratio/mean": 1.0036690326837392, "eval_sampling/importance_sampling_ratio/max": 1.3942875678722675, "eval_entropy": 0.17351046003974402, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.33899185634576356, "eval_reward_meter_mean": 0.6660121427132533, "eval_reward_meter_std": 0.37975076070198643, "eval_reward_count_adherence_mean": 0.7424245018225449, "eval_reward_count_adherence_std": 0.2582179422561939, "eval_reward_arabic_clean_mean": 0.9711538461538461, "eval_reward_arabic_clean_std": 0.08158924258672275, "eval_reward_repeat_penalty_mean": 0.638540084545429, "eval_reward_repeat_penalty_std": 0.270676647241299, "eval_reward_total_composite_mean": 0.33899185634576356, "eval_reward_total_composite_std": 0.3308297275350644} {"timestamp_utc": "2026-04-11T22:51:55Z", "mode": "train", "global_step": 751, "epoch": 0.030164276820500463, "loss": 0.0024, "grad_norm": 5.31432580947876, "learning_rate": 7.727272727272727e-06, "num_tokens": 1658395.0, "completions/mean_length": 59.25, "completions/min_length": 59.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.25, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9893741607666016, "rewards/meter/std": 0.007569636683911085, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9893741607666016, "rewards/total_composite/std": 0.007569636683911085, "reward": 0.9893741607666016, "reward_std": 0.007569642271846533, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013296845369040966, "sampling/sampling_logp_difference/max": 0.9290802478790283, "sampling/importance_sampling_ratio/min": 0.3949167728424072, "sampling/importance_sampling_ratio/mean": 1.0002750158309937, "sampling/importance_sampling_ratio/max": 1.868045449256897, "entropy": 0.052004152443259954, "clip_ratio/low_mean": 0.006320621585473418, "clip_ratio/low_min": 0.006320621585473418, "clip_ratio/high_mean": 0.008403955027461052, "clip_ratio/high_max": 0.008403955027461052, "clip_ratio/region_mean": 0.01472457661293447, "reward_total_mean": 0.9893741607666016, "reward_meter_mean": 0.9893741607666016, "reward_meter_std": 0.007569636683911085, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9893741607666016, "reward_total_composite_std": 0.007569636683911085} {"timestamp_utc": "2026-04-11T22:52:00Z", "mode": "train", "global_step": 752, "epoch": 0.030204442302285417, "loss": 0.0051, "grad_norm": 4.07772970199585, "learning_rate": 7.724242424242424e-06, "num_tokens": 1660117.0, "completions/mean_length": 59.25, "completions/min_length": 59.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.25, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9949396848678589, "rewards/meter/std": 0.00028474157443270087, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9534660577774048, "rewards/total_composite/std": 0.11713220924139023, "reward": 0.9534660577774048, "reward_std": 0.11713218688964844, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009837880730628967, "sampling/sampling_logp_difference/max": 1.2378616333007812, "sampling/importance_sampling_ratio/min": 0.2900037169456482, "sampling/importance_sampling_ratio/mean": 1.0019704103469849, "sampling/importance_sampling_ratio/max": 1.3358639478683472, "entropy": 0.03427604655735195, "clip_ratio/low_mean": 0.0020833334419876337, "clip_ratio/low_min": 0.0020833334419876337, "clip_ratio/high_mean": 0.006320621585473418, "clip_ratio/high_max": 0.006320621585473418, "clip_ratio/region_mean": 0.008403955027461052, "reward_total_mean": 0.9534660577774048, "reward_meter_mean": 0.9949396848678589, "reward_meter_std": 0.00028474157443270087, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.9534660577774048, "reward_total_composite_std": 0.11713220924139023} {"timestamp_utc": "2026-04-11T22:52:04Z", "mode": "train", "global_step": 753, "epoch": 0.03024460778407037, "loss": -0.0196, "grad_norm": 14.107855796813965, "learning_rate": 7.721212121212122e-06, "num_tokens": 1661428.0, "completions/mean_length": 35.875, "completions/min_length": 34.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.875, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.9349486827850342, "rewards/meter/std": 0.1060987114906311, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9349486827850342, "rewards/total_composite/std": 0.1060987114906311, "reward": 0.9349486827850342, "reward_std": 0.1060987114906311, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07551710307598114, "sampling/sampling_logp_difference/max": 1.5768651962280273, "sampling/importance_sampling_ratio/min": 0.20662181079387665, "sampling/importance_sampling_ratio/mean": 1.0009469985961914, "sampling/importance_sampling_ratio/max": 1.6115726232528687, "entropy": 0.6193088293075562, "clip_ratio/low_mean": 0.017439668532460928, "clip_ratio/low_min": 0.017439668532460928, "clip_ratio/high_mean": 0.04496192745864391, "clip_ratio/high_max": 0.04496192745864391, "clip_ratio/region_mean": 0.06240159599110484, "reward_total_mean": 0.9349486827850342, "reward_meter_mean": 0.9349486827850342, "reward_meter_std": 0.1060987114906311, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9349486827850342, "reward_total_composite_std": 0.1060987114906311} {"timestamp_utc": "2026-04-11T22:52:09Z", "mode": "train", "global_step": 754, "epoch": 0.030284773265855325, "loss": 0.0091, "grad_norm": 2.418649673461914, "learning_rate": 7.718181818181819e-06, "num_tokens": 1663080.0, "completions/mean_length": 41.5, "completions/min_length": 41.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.5, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9943797588348389, "rewards/meter/std": 0.0011224248446524143, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9943797588348389, "rewards/total_composite/std": 0.0011224248446524143, "reward": 0.9943797588348389, "reward_std": 0.0011224271729588509, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030370449647307396, "sampling/sampling_logp_difference/max": 1.4918346405029297, "sampling/importance_sampling_ratio/min": 0.2249595671892166, "sampling/importance_sampling_ratio/mean": 0.9925946593284607, "sampling/importance_sampling_ratio/max": 1.2741270065307617, "entropy": 0.11915541160851717, "clip_ratio/low_mean": 0.008928571594879031, "clip_ratio/low_min": 0.008928571594879031, "clip_ratio/high_mean": 0.018074912950396538, "clip_ratio/high_max": 0.018074912950396538, "clip_ratio/region_mean": 0.02700348454527557, "reward_total_mean": 0.9943797588348389, "reward_meter_mean": 0.9943797588348389, "reward_meter_std": 0.0011224248446524143, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9943797588348389, "reward_total_composite_std": 0.0011224248446524143} {"timestamp_utc": "2026-04-11T22:52:13Z", "mode": "train", "global_step": 755, "epoch": 0.03032493874764028, "loss": -0.0018, "grad_norm": 5.144253730773926, "learning_rate": 7.715151515151516e-06, "num_tokens": 1664616.0, "completions/mean_length": 41.0, "completions/min_length": 41.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.0, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.9944278001785278, "rewards/meter/std": 0.0012945194030180573, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9944278001785278, "rewards/total_composite/std": 0.0012945194030180573, "reward": 0.9944278001785278, "reward_std": 0.0012945224298164248, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014835118316113949, "sampling/sampling_logp_difference/max": 0.617079496383667, "sampling/importance_sampling_ratio/min": 0.5395178198814392, "sampling/importance_sampling_ratio/mean": 1.0021907091140747, "sampling/importance_sampling_ratio/max": 1.2102103233337402, "entropy": 0.10165042616426945, "clip_ratio/low_mean": 0.0030487803742289543, "clip_ratio/low_min": 0.0030487803742289543, "clip_ratio/high_mean": 0.012195121496915817, "clip_ratio/high_max": 0.012195121496915817, "clip_ratio/region_mean": 0.015243901871144772, "reward_total_mean": 0.9944278001785278, "reward_meter_mean": 0.9944278001785278, "reward_meter_std": 0.0012945194030180573, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9944278001785278, "reward_total_composite_std": 0.0012945194030180573} {"timestamp_utc": "2026-04-11T22:52:19Z", "mode": "train", "global_step": 756, "epoch": 0.030365104229425233, "loss": -0.0099, "grad_norm": 2.4242031574249268, "learning_rate": 7.712121212121213e-06, "num_tokens": 1666878.0, "completions/mean_length": 113.75, "completions/min_length": 111.0, "completions/max_length": 120.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 113.75, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.9662827253341675, "rewards/meter/std": 0.009305375628173351, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6500000357627869, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.6285786032676697, "rewards/total_composite/std": 0.0941418930888176, "reward": 0.6285786032676697, "reward_std": 0.09414192289113998, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011911381967365742, "sampling/sampling_logp_difference/max": 1.6042141914367676, "sampling/importance_sampling_ratio/min": 0.20104746520519257, "sampling/importance_sampling_ratio/mean": 1.0027672052383423, "sampling/importance_sampling_ratio/max": 1.911029577255249, "entropy": 0.0643842932768166, "clip_ratio/low_mean": 0.005512091098353267, "clip_ratio/low_min": 0.005512091098353267, "clip_ratio/high_mean": 0.005401917500421405, "clip_ratio/high_max": 0.005401917500421405, "clip_ratio/region_mean": 0.010914008598774672, "reward_total_mean": 0.6285786032676697, "reward_meter_mean": 0.9662827253341675, "reward_meter_std": 0.009305375628173351, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6500000357627869, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.6285786032676697, "reward_total_composite_std": 0.0941418930888176} {"timestamp_utc": "2026-04-11T22:52:26Z", "mode": "train", "global_step": 757, "epoch": 0.030405269711210187, "loss": 0.0259, "grad_norm": 1.2799259424209595, "learning_rate": 7.709090909090909e-06, "num_tokens": 1670368.0, "completions/mean_length": 243.25, "completions/min_length": 217.0, "completions/max_length": 267.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 243.25, "completions/min_terminated_length": 217.0, "completions/max_terminated_length": 267.0, "rewards/meter/mean": 0.9969884157180786, "rewards/meter/std": 0.000910833477973938, "rewards/count_adherence/mean": 0.6500000357627869, "rewards/count_adherence/std": 0.09258200973272324, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4187062978744507, "rewards/repeat_penalty/std": 0.23771634697914124, "rewards/total_composite/mean": 0.2822417914867401, "rewards/total_composite/std": 0.18207497894763947, "reward": 0.2822417914867401, "reward_std": 0.18207496404647827, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00974962953478098, "sampling/sampling_logp_difference/max": 1.2791423797607422, "sampling/importance_sampling_ratio/min": 0.27827587723731995, "sampling/importance_sampling_ratio/mean": 1.0003676414489746, "sampling/importance_sampling_ratio/max": 1.5265671014785767, "entropy": 0.06725997012108564, "clip_ratio/low_mean": 0.0010245901066809893, "clip_ratio/low_min": 0.0010245901066809893, "clip_ratio/high_mean": 0.007282168400706723, "clip_ratio/high_max": 0.007282168400706723, "clip_ratio/region_mean": 0.008306758507387713, "reward_total_mean": 0.2822417914867401, "reward_meter_mean": 0.9969884157180786, "reward_meter_std": 0.000910833477973938, "reward_count_adherence_mean": 0.6500000357627869, "reward_count_adherence_std": 0.09258200973272324, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4187062978744507, "reward_repeat_penalty_std": 0.23771634697914124, "reward_total_composite_mean": 0.2822417914867401, "reward_total_composite_std": 0.18207497894763947} {"timestamp_utc": "2026-04-11T22:52:37Z", "mode": "train", "global_step": 758, "epoch": 0.03044543519299514, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.706060606060606e-06, "num_tokens": 1672112.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.8747541308403015, "rewards/meter/std": 0.1777074635028839, "rewards/count_adherence/mean": 0.15625, "rewards/count_adherence/std": 0.11080066114664078, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.42480412125587463, "rewards/repeat_penalty/std": 0.2929826080799103, "rewards/total_composite/mean": 0.059182293713092804, "rewards/total_composite/std": 0.06593845039606094, "reward": 0.059182293713092804, "reward_std": 0.06593845039606094, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.059182293713092804, "reward_meter_mean": 0.8747541308403015, "reward_meter_std": 0.1777074635028839, "reward_count_adherence_mean": 0.15625, "reward_count_adherence_std": 0.11080066114664078, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.42480412125587463, "reward_repeat_penalty_std": 0.2929826080799103, "reward_total_composite_mean": 0.059182293713092804, "reward_total_composite_std": 0.06593845039606094} {"timestamp_utc": "2026-04-11T22:52:42Z", "mode": "train", "global_step": 759, "epoch": 0.030485600674780094, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.703030303030304e-06, "num_tokens": 1674144.0, "completions/mean_length": 87.0, "completions/min_length": 87.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.0, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.9949710369110107, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6000000238418579, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5969825983047485, "rewards/total_composite/std": 0.0, "reward": 0.5969825983047485, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.001654994674026966, "sampling/sampling_logp_difference/max": 0.1641908884048462, "sampling/importance_sampling_ratio/min": 0.848580002784729, "sampling/importance_sampling_ratio/mean": 1.0006695985794067, "sampling/importance_sampling_ratio/max": 1.0613410472869873, "entropy": 0.014668526826426387, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.5969825983047485, "reward_meter_mean": 0.9949710369110107, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6000000238418579, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5969825983047485, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-11T22:52:46Z", "mode": "train", "global_step": 760, "epoch": 0.03052576615656505, "loss": -0.0037, "grad_norm": 4.493424415588379, "learning_rate": 7.7e-06, "num_tokens": 1675945.0, "completions/mean_length": 54.125, "completions/min_length": 54.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.125, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9517734050750732, "rewards/meter/std": 0.0011382178636267781, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7083333730697632, "rewards/repeat_penalty/std": 0.11785111576318741, "rewards/total_composite/mean": 0.6740554571151733, "rewards/total_composite/std": 0.11107677966356277, "reward": 0.6740554571151733, "reward_std": 0.11107677221298218, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005725136026740074, "sampling/sampling_logp_difference/max": 0.4777810573577881, "sampling/importance_sampling_ratio/min": 0.7561957836151123, "sampling/importance_sampling_ratio/mean": 1.0033941268920898, "sampling/importance_sampling_ratio/max": 1.612492322921753, "entropy": 0.042751661501824856, "clip_ratio/low_mean": 0.004629629664123058, "clip_ratio/low_min": 0.004629629664123058, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.004629629664123058, "reward_total_mean": 0.6740554571151733, "reward_meter_mean": 0.9517734050750732, "reward_meter_std": 0.0011382178636267781, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7083333730697632, "reward_repeat_penalty_std": 0.11785111576318741, "reward_total_composite_mean": 0.6740554571151733, "reward_total_composite_std": 0.11107677966356277} {"timestamp_utc": "2026-04-11T22:52:51Z", "mode": "train", "global_step": 761, "epoch": 0.030565931638350002, "loss": -0.0023, "grad_norm": 0.8903390765190125, "learning_rate": 7.696969696969696e-06, "num_tokens": 1677747.0, "completions/mean_length": 58.25, "completions/min_length": 58.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.25, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9949068427085876, "rewards/meter/std": 0.0007968654972501099, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7083333730697632, "rewards/repeat_penalty/std": 0.11785111576318741, "rewards/total_composite/mean": 0.7047345042228699, "rewards/total_composite/std": 0.1173337996006012, "reward": 0.7047345042228699, "reward_std": 0.1173337996006012, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011640225537121296, "sampling/sampling_logp_difference/max": 1.0540056228637695, "sampling/importance_sampling_ratio/min": 0.3485388159751892, "sampling/importance_sampling_ratio/mean": 1.0014004707336426, "sampling/importance_sampling_ratio/max": 1.6007080078125, "entropy": 0.05586709058843553, "clip_ratio/low_mean": 0.006428988883271813, "clip_ratio/low_min": 0.006428988883271813, "clip_ratio/high_mean": 0.0021186440717428923, "clip_ratio/high_max": 0.0021186440717428923, "clip_ratio/region_mean": 0.008547632955014706, "reward_total_mean": 0.7047345042228699, "reward_meter_mean": 0.9949068427085876, "reward_meter_std": 0.0007968654972501099, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7083333730697632, "reward_repeat_penalty_std": 0.11785111576318741, "reward_total_composite_mean": 0.7047345042228699, "reward_total_composite_std": 0.1173337996006012} {"timestamp_utc": "2026-04-11T22:52:56Z", "mode": "train", "global_step": 762, "epoch": 0.030606097120134956, "loss": 0.024, "grad_norm": 5.6463236808776855, "learning_rate": 7.693939393939395e-06, "num_tokens": 1679613.0, "completions/mean_length": 78.25, "completions/min_length": 75.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.25, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.8964411020278931, "rewards/meter/std": 0.12688924372196198, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6750000715255737, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.6016985774040222, "rewards/total_composite/std": 0.1109333410859108, "reward": 0.6016985774040222, "reward_std": 0.11093335598707199, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02804330736398697, "sampling/sampling_logp_difference/max": 0.9042620658874512, "sampling/importance_sampling_ratio/min": 0.4048405587673187, "sampling/importance_sampling_ratio/mean": 1.0003811120986938, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11404913989827037, "clip_ratio/low_mean": 0.012610982405021787, "clip_ratio/low_min": 0.012610982405021787, "clip_ratio/high_mean": 0.006666666595265269, "clip_ratio/high_max": 0.006666666595265269, "clip_ratio/region_mean": 0.019277649000287056, "reward_total_mean": 0.6016985774040222, "reward_meter_mean": 0.8964411020278931, "reward_meter_std": 0.12688924372196198, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6750000715255737, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.6016985774040222, "reward_total_composite_std": 0.1109333410859108} {"timestamp_utc": "2026-04-11T22:53:06Z", "mode": "train", "global_step": 763, "epoch": 0.03064626260191991, "loss": 0.0481, "grad_norm": 1.2447540760040283, "learning_rate": 7.690909090909091e-06, "num_tokens": 1684742.0, "completions/mean_length": 438.125, "completions/min_length": 391.0, "completions/max_length": 498.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 438.125, "completions/min_terminated_length": 391.0, "completions/max_terminated_length": 498.0, "rewards/meter/mean": 0.9878873825073242, "rewards/meter/std": 0.019713442772626877, "rewards/count_adherence/mean": 0.2142857164144516, "rewards/count_adherence/std": 0.07636035978794098, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5576170682907104, "rewards/repeat_penalty/std": 0.0946657732129097, "rewards/total_composite/mean": 0.12146620452404022, "rewards/total_composite/std": 0.05504889413714409, "reward": 0.12146620452404022, "reward_std": 0.05504889413714409, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010950884781777859, "sampling/sampling_logp_difference/max": 1.4895976781845093, "sampling/importance_sampling_ratio/min": 0.22546334564685822, "sampling/importance_sampling_ratio/mean": 1.0031057596206665, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07848789915442467, "clip_ratio/low_mean": 0.005670643062330782, "clip_ratio/low_min": 0.005670643062330782, "clip_ratio/high_mean": 0.005744358117226511, "clip_ratio/high_max": 0.005744358117226511, "clip_ratio/region_mean": 0.011415001179557294, "reward_total_mean": 0.12146620452404022, "reward_meter_mean": 0.9878873825073242, "reward_meter_std": 0.019713442772626877, "reward_count_adherence_mean": 0.2142857164144516, "reward_count_adherence_std": 0.07636035978794098, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5576170682907104, "reward_repeat_penalty_std": 0.0946657732129097, "reward_total_composite_mean": 0.12146620452404022, "reward_total_composite_std": 0.05504889413714409} {"timestamp_utc": "2026-04-11T22:53:11Z", "mode": "train", "global_step": 764, "epoch": 0.030686428083704864, "loss": 0.0189, "grad_norm": 5.640087127685547, "learning_rate": 7.687878787878788e-06, "num_tokens": 1686539.0, "completions/mean_length": 63.625, "completions/min_length": 60.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.625, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.874701976776123, "rewards/meter/std": 0.18508003652095795, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.7850605845451355, "rewards/total_composite/std": 0.2699912190437317, "reward": 0.7850605845451355, "reward_std": 0.2699912190437317, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.032904475927352905, "sampling/sampling_logp_difference/max": 1.8050861358642578, "sampling/importance_sampling_ratio/min": 0.16446030139923096, "sampling/importance_sampling_ratio/mean": 0.9924401044845581, "sampling/importance_sampling_ratio/max": 1.438651442527771, "entropy": 0.1180117940530181, "clip_ratio/low_mean": 0.005871212342754006, "clip_ratio/low_min": 0.005871212342754006, "clip_ratio/high_mean": 0.02714023506268859, "clip_ratio/high_max": 0.02714023506268859, "clip_ratio/region_mean": 0.033011447405442595, "reward_total_mean": 0.7850605845451355, "reward_meter_mean": 0.874701976776123, "reward_meter_std": 0.18508003652095795, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.17251639068126678, "reward_total_composite_mean": 0.7850605845451355, "reward_total_composite_std": 0.2699912190437317} {"timestamp_utc": "2026-04-11T22:53:16Z", "mode": "train", "global_step": 765, "epoch": 0.030726593565489818, "loss": 0.0008, "grad_norm": 4.968373775482178, "learning_rate": 7.684848484848485e-06, "num_tokens": 1688548.0, "completions/mean_length": 87.125, "completions/min_length": 87.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.125, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.9950021505355835, "rewards/meter/std": 8.806584810372442e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6500000357627869, "rewards/repeat_penalty/std": 0.1414213478565216, "rewards/total_composite/mean": 0.6467622518539429, "rewards/total_composite/std": 0.14079822599887848, "reward": 0.6467622518539429, "reward_std": 0.14079821109771729, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004592791199684143, "sampling/sampling_logp_difference/max": 1.7399308681488037, "sampling/importance_sampling_ratio/min": 0.1755325347185135, "sampling/importance_sampling_ratio/mean": 0.9999979138374329, "sampling/importance_sampling_ratio/max": 1.767101764678955, "entropy": 0.008648848335724324, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0028409091755747795, "clip_ratio/high_max": 0.0028409091755747795, "clip_ratio/region_mean": 0.0028409091755747795, "reward_total_mean": 0.6467622518539429, "reward_meter_mean": 0.9950021505355835, "reward_meter_std": 8.806584810372442e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6500000357627869, "reward_repeat_penalty_std": 0.1414213478565216, "reward_total_composite_mean": 0.6467622518539429, "reward_total_composite_std": 0.14079822599887848} {"timestamp_utc": "2026-04-11T22:53:21Z", "mode": "train", "global_step": 766, "epoch": 0.030766759047274772, "loss": 0.0014, "grad_norm": 10.79392147064209, "learning_rate": 7.681818181818183e-06, "num_tokens": 1690236.0, "completions/mean_length": 52.0, "completions/min_length": 51.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.0, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.8247180581092834, "rewards/meter/std": 0.3005145490169525, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8247180581092834, "rewards/total_composite/std": 0.3005145490169525, "reward": 0.8247180581092834, "reward_std": 0.3005145490169525, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04232970252633095, "sampling/sampling_logp_difference/max": 1.6089832782745361, "sampling/importance_sampling_ratio/min": 0.2000909447669983, "sampling/importance_sampling_ratio/mean": 0.997559666633606, "sampling/importance_sampling_ratio/max": 1.6643987894058228, "entropy": 0.19545143470168114, "clip_ratio/low_mean": 0.004807692486792803, "clip_ratio/low_min": 0.004807692486792803, "clip_ratio/high_mean": 0.03126057400368154, "clip_ratio/high_max": 0.03126057400368154, "clip_ratio/region_mean": 0.03606826649047434, "reward_total_mean": 0.8247180581092834, "reward_meter_mean": 0.8247180581092834, "reward_meter_std": 0.3005145490169525, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8247180581092834, "reward_total_composite_std": 0.3005145490169525} {"timestamp_utc": "2026-04-11T22:53:26Z", "mode": "train", "global_step": 767, "epoch": 0.030806924529059726, "loss": 0.0103, "grad_norm": 12.172296524047852, "learning_rate": 7.678787878787878e-06, "num_tokens": 1692065.0, "completions/mean_length": 72.625, "completions/min_length": 67.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.625, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.3632194995880127, "rewards/meter/std": 0.32341283559799194, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.36088046431541443, "rewards/total_composite/std": 0.3263150453567505, "reward": 0.36088046431541443, "reward_std": 0.3263150453567505, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07868395745754242, "sampling/sampling_logp_difference/max": 5.154880046844482, "sampling/importance_sampling_ratio/min": 0.005771172232925892, "sampling/importance_sampling_ratio/mean": 0.9972518086433411, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.46072521805763245, "clip_ratio/low_mean": 0.03220835281535983, "clip_ratio/low_min": 0.03220835281535983, "clip_ratio/high_mean": 0.026226354064419866, "clip_ratio/high_max": 0.026226354064419866, "clip_ratio/region_mean": 0.058434706879779696, "reward_total_mean": 0.36088046431541443, "reward_meter_mean": 0.3632194995880127, "reward_meter_std": 0.32341283559799194, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.36088046431541443, "reward_total_composite_std": 0.3263150453567505} {"timestamp_utc": "2026-04-11T22:53:31Z", "mode": "train", "global_step": 768, "epoch": 0.03084709001084468, "loss": -0.0095, "grad_norm": 2.0709922313690186, "learning_rate": 7.675757575757577e-06, "num_tokens": 1694344.0, "completions/mean_length": 107.875, "completions/min_length": 104.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.875, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.7045435905456543, "rewards/meter/std": 0.13543730974197388, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5636348724365234, "rewards/total_composite/std": 0.10834985971450806, "reward": 0.5636348724365234, "reward_std": 0.10834983736276627, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010335095226764679, "sampling/sampling_logp_difference/max": 0.9596023559570312, "sampling/importance_sampling_ratio/min": 0.38304516673088074, "sampling/importance_sampling_ratio/mean": 1.0031026601791382, "sampling/importance_sampling_ratio/max": 1.4763460159301758, "entropy": 0.09003173373639584, "clip_ratio/low_mean": 0.0059091257862746716, "clip_ratio/low_min": 0.0059091257862746716, "clip_ratio/high_mean": 0.0022935778833925724, "clip_ratio/high_max": 0.0022935778833925724, "clip_ratio/region_mean": 0.008202703669667244, "reward_total_mean": 0.5636348724365234, "reward_meter_mean": 0.7045435905456543, "reward_meter_std": 0.13543730974197388, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5636348724365234, "reward_total_composite_std": 0.10834985971450806} {"timestamp_utc": "2026-04-11T22:53:37Z", "mode": "train", "global_step": 769, "epoch": 0.030887255492629634, "loss": 0.0714, "grad_norm": 7.707916736602783, "learning_rate": 7.672727272727273e-06, "num_tokens": 1696973.0, "completions/mean_length": 143.625, "completions/min_length": 126.0, "completions/max_length": 172.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 143.625, "completions/min_terminated_length": 126.0, "completions/max_terminated_length": 172.0, "rewards/meter/mean": 0.7441333532333374, "rewards/meter/std": 0.4554755687713623, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6174242496490479, "rewards/repeat_penalty/std": 0.1036364957690239, "rewards/total_composite/mean": 0.4105571210384369, "rewards/total_composite/std": 0.2803230583667755, "reward": 0.4105571210384369, "reward_std": 0.2803230583667755, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03509281575679779, "sampling/sampling_logp_difference/max": 4.862173557281494, "sampling/importance_sampling_ratio/min": 0.0077336556278169155, "sampling/importance_sampling_ratio/mean": 0.9966462850570679, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0793102509342134, "clip_ratio/low_mean": 0.009834110038354993, "clip_ratio/low_min": 0.009834110038354993, "clip_ratio/high_mean": 0.0102224723668769, "clip_ratio/high_max": 0.0102224723668769, "clip_ratio/region_mean": 0.020056582405231893, "reward_total_mean": 0.4105571210384369, "reward_meter_mean": 0.7441333532333374, "reward_meter_std": 0.4554755687713623, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6174242496490479, "reward_repeat_penalty_std": 0.1036364957690239, "reward_total_composite_mean": 0.4105571210384369, "reward_total_composite_std": 0.2803230583667755} {"timestamp_utc": "2026-04-11T22:53:47Z", "mode": "train", "global_step": 770, "epoch": 0.030927420974414588, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.66969696969697e-06, "num_tokens": 1698965.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.8939934968948364, "rewards/meter/std": 0.13721102476119995, "rewards/count_adherence/mean": 0.7828947305679321, "rewards/count_adherence/std": 0.03373000770807266, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5402884483337402, "rewards/repeat_penalty/std": 0.2157072126865387, "rewards/total_composite/mean": 0.3744324743747711, "rewards/total_composite/std": 0.17441345751285553, "reward": 0.3744324743747711, "reward_std": 0.17441345751285553, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.3744324743747711, "reward_meter_mean": 0.8939934968948364, "reward_meter_std": 0.13721102476119995, "reward_count_adherence_mean": 0.7828947305679321, "reward_count_adherence_std": 0.03373000770807266, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5402884483337402, "reward_repeat_penalty_std": 0.2157072126865387, "reward_total_composite_mean": 0.3744324743747711, "reward_total_composite_std": 0.17441345751285553} {"timestamp_utc": "2026-04-11T22:53:54Z", "mode": "train", "global_step": 771, "epoch": 0.030967586456199542, "loss": 0.0526, "grad_norm": 4.9908342361450195, "learning_rate": 7.666666666666667e-06, "num_tokens": 1702409.0, "completions/mean_length": 230.5, "completions/min_length": 180.0, "completions/max_length": 263.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 230.5, "completions/min_terminated_length": 180.0, "completions/max_terminated_length": 263.0, "rewards/meter/mean": 0.7852883338928223, "rewards/meter/std": 0.3382474482059479, "rewards/count_adherence/mean": 0.8250000476837158, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5688130855560303, "rewards/repeat_penalty/std": 0.20688475668430328, "rewards/total_composite/mean": 0.3491661250591278, "rewards/total_composite/std": 0.22600221633911133, "reward": 0.3491661250591278, "reward_std": 0.22600221633911133, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018919790163636208, "sampling/sampling_logp_difference/max": 1.5345475673675537, "sampling/importance_sampling_ratio/min": 0.21555320918560028, "sampling/importance_sampling_ratio/mean": 1.0030508041381836, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.13555688876658678, "clip_ratio/low_mean": 0.004294316866435111, "clip_ratio/low_min": 0.004294316866435111, "clip_ratio/high_mean": 0.010813763190526515, "clip_ratio/high_max": 0.010813763190526515, "clip_ratio/region_mean": 0.015108080056961626, "reward_total_mean": 0.3491661250591278, "reward_meter_mean": 0.7852883338928223, "reward_meter_std": 0.3382474482059479, "reward_count_adherence_mean": 0.8250000476837158, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5688130855560303, "reward_repeat_penalty_std": 0.20688475668430328, "reward_total_composite_mean": 0.3491661250591278, "reward_total_composite_std": 0.22600221633911133} {"timestamp_utc": "2026-04-11T22:54:00Z", "mode": "train", "global_step": 772, "epoch": 0.031007751937984496, "loss": -0.1232, "grad_norm": 5.677700042724609, "learning_rate": 7.663636363636364e-06, "num_tokens": 1704516.0, "completions/mean_length": 99.375, "completions/min_length": 90.0, "completions/max_length": 123.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.375, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.9852504730224609, "rewards/meter/std": 0.008375532925128937, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.1157275140285492, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7785714864730835, "rewards/repeat_penalty/std": 0.039677999913692474, "rewards/total_composite/mean": 0.6195021271705627, "rewards/total_composite/std": 0.05528084933757782, "reward": 0.6195021271705627, "reward_std": 0.05528085306286812, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020249057561159134, "sampling/sampling_logp_difference/max": 1.5644612312316895, "sampling/importance_sampling_ratio/min": 0.2092006951570511, "sampling/importance_sampling_ratio/mean": 1.0006357431411743, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08717024885118008, "clip_ratio/low_mean": 0.015005706925876439, "clip_ratio/low_min": 0.015005706925876439, "clip_ratio/high_mean": 0.006097560748457909, "clip_ratio/high_max": 0.006097560748457909, "clip_ratio/region_mean": 0.021103267674334347, "reward_total_mean": 0.6195021271705627, "reward_meter_mean": 0.9852504730224609, "reward_meter_std": 0.008375532925128937, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.1157275140285492, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7785714864730835, "reward_repeat_penalty_std": 0.039677999913692474, "reward_total_composite_mean": 0.6195021271705627, "reward_total_composite_std": 0.05528084933757782} {"timestamp_utc": "2026-04-11T22:54:09Z", "mode": "train", "global_step": 773, "epoch": 0.03104791741976945, "loss": 0.0245, "grad_norm": 5.519010066986084, "learning_rate": 7.660606060606062e-06, "num_tokens": 1709201.0, "completions/mean_length": 409.625, "completions/min_length": 346.0, "completions/max_length": 460.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 409.625, "completions/min_terminated_length": 346.0, "completions/max_terminated_length": 460.0, "rewards/meter/mean": 0.9907213449478149, "rewards/meter/std": 0.006963523104786873, "rewards/count_adherence/mean": 0.3571428656578064, "rewards/count_adherence/std": 0.07636035233736038, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.45148104429244995, "rewards/repeat_penalty/std": 0.23929926753044128, "rewards/total_composite/mean": 0.17074038088321686, "rewards/total_composite/std": 0.10653632879257202, "reward": 0.17074038088321686, "reward_std": 0.10653632134199142, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02135242149233818, "sampling/sampling_logp_difference/max": 5.103124141693115, "sampling/importance_sampling_ratio/min": 0.006077729165554047, "sampling/importance_sampling_ratio/mean": 0.9987974166870117, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07645870675332844, "clip_ratio/low_mean": 0.003349777136463672, "clip_ratio/low_min": 0.003349777136463672, "clip_ratio/high_mean": 0.008041649358347058, "clip_ratio/high_max": 0.008041649358347058, "clip_ratio/region_mean": 0.01139142649481073, "reward_total_mean": 0.17074038088321686, "reward_meter_mean": 0.9907213449478149, "reward_meter_std": 0.006963523104786873, "reward_count_adherence_mean": 0.3571428656578064, "reward_count_adherence_std": 0.07636035233736038, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.45148104429244995, "reward_repeat_penalty_std": 0.23929926753044128, "reward_total_composite_mean": 0.17074038088321686, "reward_total_composite_std": 0.10653632879257202} {"timestamp_utc": "2026-04-11T22:54:14Z", "mode": "train", "global_step": 774, "epoch": 0.031088082901554404, "loss": -0.0134, "grad_norm": 6.2788872718811035, "learning_rate": 7.657575757575757e-06, "num_tokens": 1711058.0, "completions/mean_length": 69.125, "completions/min_length": 65.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.125, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.7258920669555664, "rewards/meter/std": 0.40674278140068054, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7258920669555664, "rewards/total_composite/std": 0.40674278140068054, "reward": 0.7258920669555664, "reward_std": 0.40674278140068054, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05154965817928314, "sampling/sampling_logp_difference/max": 2.2386436462402344, "sampling/importance_sampling_ratio/min": 0.1066029891371727, "sampling/importance_sampling_ratio/mean": 1.0082036256790161, "sampling/importance_sampling_ratio/max": 1.882003903388977, "entropy": 0.37627044692635536, "clip_ratio/low_mean": 0.005654420121572912, "clip_ratio/low_min": 0.005654420121572912, "clip_ratio/high_mean": 0.022313665016554296, "clip_ratio/high_max": 0.022313665016554296, "clip_ratio/region_mean": 0.027968085138127208, "reward_total_mean": 0.7258920669555664, "reward_meter_mean": 0.7258920669555664, "reward_meter_std": 0.40674278140068054, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7258920669555664, "reward_total_composite_std": 0.40674278140068054} {"timestamp_utc": "2026-04-11T22:54:20Z", "mode": "train", "global_step": 775, "epoch": 0.031128248383339358, "loss": 0.0636, "grad_norm": 2.807114601135254, "learning_rate": 7.654545454545456e-06, "num_tokens": 1713733.0, "completions/mean_length": 168.375, "completions/min_length": 147.0, "completions/max_length": 188.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 168.375, "completions/min_terminated_length": 147.0, "completions/max_terminated_length": 188.0, "rewards/meter/mean": 0.9743660688400269, "rewards/meter/std": 0.04519447684288025, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6060605645179749, "rewards/repeat_penalty/std": 0.13551926612854004, "rewards/total_composite/mean": 0.5430476665496826, "rewards/total_composite/std": 0.16461656987667084, "reward": 0.5430476665496826, "reward_std": 0.16461656987667084, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026703810319304466, "sampling/sampling_logp_difference/max": 12.626859664916992, "sampling/importance_sampling_ratio/min": 3.2826496862981003e-06, "sampling/importance_sampling_ratio/mean": 0.9989210367202759, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06733021000400186, "clip_ratio/low_mean": 0.00418766331858933, "clip_ratio/low_min": 0.00418766331858933, "clip_ratio/high_mean": 0.007139897206798196, "clip_ratio/high_max": 0.007139897206798196, "clip_ratio/region_mean": 0.011327560525387526, "reward_total_mean": 0.5430476665496826, "reward_meter_mean": 0.9743660688400269, "reward_meter_std": 0.04519447684288025, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6060605645179749, "reward_repeat_penalty_std": 0.13551926612854004, "reward_total_composite_mean": 0.5430476665496826, "reward_total_composite_std": 0.16461656987667084} {"timestamp_utc": "2026-04-11T22:54:25Z", "mode": "train", "global_step": 776, "epoch": 0.03116841386512431, "loss": 0.0012, "grad_norm": 7.4227614402771, "learning_rate": 7.651515151515152e-06, "num_tokens": 1715409.0, "completions/mean_length": 54.5, "completions/min_length": 53.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.5, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.9904471635818481, "rewards/meter/std": 0.002613567281514406, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9904471635818481, "rewards/total_composite/std": 0.002613567281514406, "reward": 0.9904471635818481, "reward_std": 0.002613575430586934, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024185476824641228, "sampling/sampling_logp_difference/max": 1.0824594497680664, "sampling/importance_sampling_ratio/min": 0.3387613296508789, "sampling/importance_sampling_ratio/mean": 0.9967302083969116, "sampling/importance_sampling_ratio/max": 1.4630638360977173, "entropy": 0.09682174911722541, "clip_ratio/low_mean": 0.009263938991352916, "clip_ratio/low_min": 0.009263938991352916, "clip_ratio/high_mean": 0.009050324792042375, "clip_ratio/high_max": 0.009050324792042375, "clip_ratio/region_mean": 0.01831426378339529, "reward_total_mean": 0.9904471635818481, "reward_meter_mean": 0.9904471635818481, "reward_meter_std": 0.002613567281514406, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9904471635818481, "reward_total_composite_std": 0.002613567281514406} {"timestamp_utc": "2026-04-11T22:54:29Z", "mode": "train", "global_step": 777, "epoch": 0.031208579346909265, "loss": 0.0046, "grad_norm": 8.215063095092773, "learning_rate": 7.648484848484849e-06, "num_tokens": 1716925.0, "completions/mean_length": 36.5, "completions/min_length": 36.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.5, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.978127121925354, "rewards/meter/std": 0.0063257296569645405, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.978127121925354, "rewards/total_composite/std": 0.0063257296569645405, "reward": 0.978127121925354, "reward_std": 0.006325736176222563, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03158137574791908, "sampling/sampling_logp_difference/max": 1.4631445407867432, "sampling/importance_sampling_ratio/min": 0.23150713741779327, "sampling/importance_sampling_ratio/mean": 1.0036948919296265, "sampling/importance_sampling_ratio/max": 1.6868826150894165, "entropy": 0.11897748988121748, "clip_ratio/low_mean": 0.013795045437291265, "clip_ratio/low_min": 0.013795045437291265, "clip_ratio/high_mean": 0.013795045204460621, "clip_ratio/high_max": 0.013795045204460621, "clip_ratio/region_mean": 0.027590090641751885, "reward_total_mean": 0.978127121925354, "reward_meter_mean": 0.978127121925354, "reward_meter_std": 0.0063257296569645405, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.978127121925354, "reward_total_composite_std": 0.0063257296569645405} {"timestamp_utc": "2026-04-11T22:54:34Z", "mode": "train", "global_step": 778, "epoch": 0.03124874482869422, "loss": 0.014, "grad_norm": 5.06115198135376, "learning_rate": 7.645454545454546e-06, "num_tokens": 1718951.0, "completions/mean_length": 72.25, "completions/min_length": 68.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.25, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9873093366622925, "rewards/meter/std": 0.01036460418254137, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9873093366622925, "rewards/total_composite/std": 0.01036460418254137, "reward": 0.9873093366622925, "reward_std": 0.010364595800638199, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03534568101167679, "sampling/sampling_logp_difference/max": 0.8904938697814941, "sampling/importance_sampling_ratio/min": 0.4104529917240143, "sampling/importance_sampling_ratio/mean": 1.0149582624435425, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.25925541296601295, "clip_ratio/low_mean": 0.02250340231694281, "clip_ratio/low_min": 0.02250340231694281, "clip_ratio/high_mean": 0.020928236190229654, "clip_ratio/high_max": 0.020928236190229654, "clip_ratio/region_mean": 0.043431638507172465, "reward_total_mean": 0.9873093366622925, "reward_meter_mean": 0.9873093366622925, "reward_meter_std": 0.01036460418254137, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9873093366622925, "reward_total_composite_std": 0.01036460418254137} {"timestamp_utc": "2026-04-11T22:54:45Z", "mode": "train", "global_step": 779, "epoch": 0.03128891031047917, "loss": -0.1238, "grad_norm": 1.4244983196258545, "learning_rate": 7.642424242424244e-06, "num_tokens": 1720483.0, "completions/mean_length": 182.5, "completions/min_length": 69.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 72.66667175292969, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.6190794110298157, "rewards/meter/std": 0.3674232065677643, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5616785287857056, "rewards/total_composite/std": 0.43257132172584534, "reward": 0.5616785287857056, "reward_std": 0.43257129192352295, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07225891947746277, "sampling/sampling_logp_difference/max": 1.2201738357543945, "sampling/importance_sampling_ratio/min": 0.2951788604259491, "sampling/importance_sampling_ratio/mean": 1.0154547691345215, "sampling/importance_sampling_ratio/max": 1.9660524129867554, "entropy": 0.38310878723859787, "clip_ratio/low_mean": 0.008561643771827221, "clip_ratio/low_min": 0.008561643771827221, "clip_ratio/high_mean": 0.030813875840976834, "clip_ratio/high_max": 0.030813875840976834, "clip_ratio/region_mean": 0.039375519612804055, "reward_total_mean": 0.5616785287857056, "reward_meter_mean": 0.6190794110298157, "reward_meter_std": 0.3674232065677643, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5616785287857056, "reward_total_composite_std": 0.43257132172584534} {"timestamp_utc": "2026-04-11T22:54:56Z", "mode": "train", "global_step": 780, "epoch": 0.03132907579226413, "loss": -0.0498, "grad_norm": 3.119354009628296, "learning_rate": 7.639393939393939e-06, "num_tokens": 1722862.0, "completions/mean_length": 174.375, "completions/min_length": 111.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 126.14286041259766, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 182.0, "rewards/meter/mean": 0.6168656349182129, "rewards/meter/std": 0.4164867103099823, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.7492559552192688, "rewards/repeat_penalty/std": 0.14768067002296448, "rewards/total_composite/mean": 0.4575069546699524, "rewards/total_composite/std": 0.3478102684020996, "reward": 0.4575069546699524, "reward_std": 0.3478102684020996, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03274521231651306, "sampling/sampling_logp_difference/max": 1.078242540359497, "sampling/importance_sampling_ratio/min": 0.34019285440444946, "sampling/importance_sampling_ratio/mean": 1.0024807453155518, "sampling/importance_sampling_ratio/max": 1.7341415882110596, "entropy": 0.24183030799031258, "clip_ratio/low_mean": 0.004807692486792803, "clip_ratio/low_min": 0.004807692486792803, "clip_ratio/high_mean": 0.029140884289518, "clip_ratio/high_max": 0.029140884289518, "clip_ratio/region_mean": 0.0339485767763108, "reward_total_mean": 0.4575069546699524, "reward_meter_mean": 0.6168656349182129, "reward_meter_std": 0.4164867103099823, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.7492559552192688, "reward_repeat_penalty_std": 0.14768067002296448, "reward_total_composite_mean": 0.4575069546699524, "reward_total_composite_std": 0.3478102684020996} {"timestamp_utc": "2026-04-11T22:55:01Z", "mode": "train", "global_step": 781, "epoch": 0.03136924127404908, "loss": -0.0116, "grad_norm": 12.43602180480957, "learning_rate": 7.636363636363638e-06, "num_tokens": 1724537.0, "completions/mean_length": 47.375, "completions/min_length": 44.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.375, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.5619980096817017, "rewards/meter/std": 0.3257886469364166, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.4856577217578888, "rewards/total_composite/std": 0.3285104036331177, "reward": 0.4856577217578888, "reward_std": 0.3285104036331177, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09212920814752579, "sampling/sampling_logp_difference/max": 1.6957062482833862, "sampling/importance_sampling_ratio/min": 0.18346960842609406, "sampling/importance_sampling_ratio/mean": 0.9970747232437134, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9118200056254864, "clip_ratio/low_mean": 0.04398810095153749, "clip_ratio/low_min": 0.04398810095153749, "clip_ratio/high_mean": 0.016590908635407686, "clip_ratio/high_max": 0.016590908635407686, "clip_ratio/region_mean": 0.060579009586945176, "reward_total_mean": 0.4856577217578888, "reward_meter_mean": 0.5619980096817017, "reward_meter_std": 0.3257886469364166, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.17251639068126678, "reward_total_composite_mean": 0.4856577217578888, "reward_total_composite_std": 0.3285104036331177} {"timestamp_utc": "2026-04-11T22:55:10Z", "mode": "train", "global_step": 782, "epoch": 0.031409406755834035, "loss": -0.1573, "grad_norm": 1.6532626152038574, "learning_rate": 7.633333333333334e-06, "num_tokens": 1726623.0, "completions/mean_length": 166.75, "completions/min_length": 111.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 117.42857360839844, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.6895759105682373, "rewards/meter/std": 0.26801931858062744, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.824999988079071, "rewards/repeat_penalty/std": 0.19820624589920044, "rewards/total_composite/mean": 0.5303109288215637, "rewards/total_composite/std": 0.28132033348083496, "reward": 0.5303109288215637, "reward_std": 0.28132033348083496, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03518048673868179, "sampling/sampling_logp_difference/max": 1.1620054244995117, "sampling/importance_sampling_ratio/min": 0.31285813450813293, "sampling/importance_sampling_ratio/mean": 1.006204605102539, "sampling/importance_sampling_ratio/max": 1.9478799104690552, "entropy": 0.2663711039349437, "clip_ratio/low_mean": 0.008460594224743545, "clip_ratio/low_min": 0.008460594224743545, "clip_ratio/high_mean": 0.013982121949084103, "clip_ratio/high_max": 0.013982121949084103, "clip_ratio/region_mean": 0.022442716173827648, "reward_total_mean": 0.5303109288215637, "reward_meter_mean": 0.6895759105682373, "reward_meter_std": 0.26801931858062744, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.824999988079071, "reward_repeat_penalty_std": 0.19820624589920044, "reward_total_composite_mean": 0.5303109288215637, "reward_total_composite_std": 0.28132033348083496} {"timestamp_utc": "2026-04-11T22:55:15Z", "mode": "train", "global_step": 783, "epoch": 0.03144957223761899, "loss": 0.0077, "grad_norm": 4.994919776916504, "learning_rate": 7.630303030303031e-06, "num_tokens": 1728611.0, "completions/mean_length": 79.5, "completions/min_length": 76.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.5, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.6204426288604736, "rewards/meter/std": 0.3357177972793579, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6204426288604736, "rewards/total_composite/std": 0.3357177972793579, "reward": 0.6204426288604736, "reward_std": 0.3357177972793579, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04477819800376892, "sampling/sampling_logp_difference/max": 1.0638208389282227, "sampling/importance_sampling_ratio/min": 0.34513458609580994, "sampling/importance_sampling_ratio/mean": 1.011016845703125, "sampling/importance_sampling_ratio/max": 1.7397865056991577, "entropy": 0.3720816671848297, "clip_ratio/low_mean": 0.017455301131121814, "clip_ratio/low_min": 0.017455301131121814, "clip_ratio/high_mean": 0.013986280770041049, "clip_ratio/high_max": 0.013986280770041049, "clip_ratio/region_mean": 0.03144158190116286, "reward_total_mean": 0.6204426288604736, "reward_meter_mean": 0.6204426288604736, "reward_meter_std": 0.3357177972793579, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6204426288604736, "reward_total_composite_std": 0.3357177972793579} {"timestamp_utc": "2026-04-11T22:55:20Z", "mode": "train", "global_step": 784, "epoch": 0.03148973771940394, "loss": 0.0052, "grad_norm": 7.485208988189697, "learning_rate": 7.627272727272727e-06, "num_tokens": 1730350.0, "completions/mean_length": 61.375, "completions/min_length": 56.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.375, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.5055776834487915, "rewards/meter/std": 0.32300832867622375, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8333333730697632, "rewards/repeat_penalty/std": 0.17817415297031403, "rewards/total_composite/mean": 0.453016996383667, "rewards/total_composite/std": 0.33306846022605896, "reward": 0.453016996383667, "reward_std": 0.33306846022605896, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.047304026782512665, "sampling/sampling_logp_difference/max": 1.7305419445037842, "sampling/importance_sampling_ratio/min": 0.17718835175037384, "sampling/importance_sampling_ratio/mean": 0.9948754906654358, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2184693105518818, "clip_ratio/low_mean": 0.018428327050060034, "clip_ratio/low_min": 0.018428327050060034, "clip_ratio/high_mean": 0.012099922401830554, "clip_ratio/high_max": 0.012099922401830554, "clip_ratio/region_mean": 0.030528249451890588, "reward_total_mean": 0.453016996383667, "reward_meter_mean": 0.5055776834487915, "reward_meter_std": 0.32300832867622375, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8333333730697632, "reward_repeat_penalty_std": 0.17817415297031403, "reward_total_composite_mean": 0.453016996383667, "reward_total_composite_std": 0.33306846022605896} {"timestamp_utc": "2026-04-11T22:55:25Z", "mode": "train", "global_step": 785, "epoch": 0.0315299032011889, "loss": 0.0036, "grad_norm": 6.059953212738037, "learning_rate": 7.6242424242424254e-06, "num_tokens": 1732524.0, "completions/mean_length": 104.75, "completions/min_length": 99.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 104.75, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.49319469928741455, "rewards/meter/std": 0.2711309790611267, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.46458256244659424, "rewards/total_composite/std": 0.2836271822452545, "reward": 0.46458256244659424, "reward_std": 0.2836271822452545, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.042036473751068115, "sampling/sampling_logp_difference/max": 1.2848865985870361, "sampling/importance_sampling_ratio/min": 0.27668195962905884, "sampling/importance_sampling_ratio/mean": 1.0086406469345093, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31474705785512924, "clip_ratio/low_mean": 0.010869022691622376, "clip_ratio/low_min": 0.010869022691622376, "clip_ratio/high_mean": 0.01767290150746703, "clip_ratio/high_max": 0.01767290150746703, "clip_ratio/region_mean": 0.028541924199089408, "reward_total_mean": 0.46458256244659424, "reward_meter_mean": 0.49319469928741455, "reward_meter_std": 0.2711309790611267, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_total_composite_mean": 0.46458256244659424, "reward_total_composite_std": 0.2836271822452545} {"timestamp_utc": "2026-04-11T22:55:34Z", "mode": "train", "global_step": 786, "epoch": 0.03157006868297385, "loss": -0.0567, "grad_norm": 2.024092435836792, "learning_rate": 7.621212121212122e-06, "num_tokens": 1737190.0, "completions/mean_length": 368.25, "completions/min_length": 322.0, "completions/max_length": 406.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 368.25, "completions/min_terminated_length": 322.0, "completions/max_terminated_length": 406.0, "rewards/meter/mean": 0.7014279365539551, "rewards/meter/std": 0.43751856684684753, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5806276798248291, "rewards/repeat_penalty/std": 0.06668182462453842, "rewards/total_composite/mean": 0.3579794764518738, "rewards/total_composite/std": 0.22696760296821594, "reward": 0.3579794764518738, "reward_std": 0.22696760296821594, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012620110996067524, "sampling/sampling_logp_difference/max": 1.491776466369629, "sampling/importance_sampling_ratio/min": 0.2249726504087448, "sampling/importance_sampling_ratio/mean": 1.0009483098983765, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07080868305638433, "clip_ratio/low_mean": 0.004607091657817364, "clip_ratio/low_min": 0.004607091657817364, "clip_ratio/high_mean": 0.0042421949619892985, "clip_ratio/high_max": 0.0042421949619892985, "clip_ratio/region_mean": 0.008849286619806662, "reward_total_mean": 0.3579794764518738, "reward_meter_mean": 0.7014279365539551, "reward_meter_std": 0.43751856684684753, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5806276798248291, "reward_repeat_penalty_std": 0.06668182462453842, "reward_total_composite_mean": 0.3579794764518738, "reward_total_composite_std": 0.22696760296821594} {"timestamp_utc": "2026-04-11T22:55:40Z", "mode": "train", "global_step": 787, "epoch": 0.031610234164758805, "loss": 0.0123, "grad_norm": 3.6839067935943604, "learning_rate": 7.618181818181819e-06, "num_tokens": 1739852.0, "completions/mean_length": 153.75, "completions/min_length": 135.0, "completions/max_length": 162.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 153.75, "completions/min_terminated_length": 135.0, "completions/max_terminated_length": 162.0, "rewards/meter/mean": 0.7929247617721558, "rewards/meter/std": 0.10911522805690765, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6818181872367859, "rewards/repeat_penalty/std": 0.06872082501649857, "rewards/total_composite/mean": 0.435272216796875, "rewards/total_composite/std": 0.09194310009479523, "reward": 0.435272216796875, "reward_std": 0.09194309264421463, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028897186741232872, "sampling/sampling_logp_difference/max": 1.9267759323120117, "sampling/importance_sampling_ratio/min": 0.14561693370342255, "sampling/importance_sampling_ratio/mean": 0.9994563460350037, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11316746287047863, "clip_ratio/low_mean": 0.01558087719604373, "clip_ratio/low_min": 0.01558087719604373, "clip_ratio/high_mean": 0.007280809339135885, "clip_ratio/high_max": 0.007280809339135885, "clip_ratio/region_mean": 0.022861686535179615, "reward_total_mean": 0.435272216796875, "reward_meter_mean": 0.7929247617721558, "reward_meter_std": 0.10911522805690765, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6818181872367859, "reward_repeat_penalty_std": 0.06872082501649857, "reward_total_composite_mean": 0.435272216796875, "reward_total_composite_std": 0.09194310009479523} {"timestamp_utc": "2026-04-11T22:55:44Z", "mode": "train", "global_step": 788, "epoch": 0.03165039964654376, "loss": -0.0051, "grad_norm": 7.856208324432373, "learning_rate": 7.6151515151515155e-06, "num_tokens": 1741588.0, "completions/mean_length": 58.0, "completions/min_length": 57.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.6469426155090332, "rewards/meter/std": 0.24868278205394745, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6469426155090332, "rewards/total_composite/std": 0.24868278205394745, "reward": 0.6469426155090332, "reward_std": 0.24868276715278625, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022744039073586464, "sampling/sampling_logp_difference/max": 0.6672461032867432, "sampling/importance_sampling_ratio/min": 0.5131197571754456, "sampling/importance_sampling_ratio/mean": 1.0034902095794678, "sampling/importance_sampling_ratio/max": 1.8282591104507446, "entropy": 0.16343743726611137, "clip_ratio/low_mean": 0.012860541231930256, "clip_ratio/low_min": 0.012860541231930256, "clip_ratio/high_mean": 0.01074189692735672, "clip_ratio/high_max": 0.01074189692735672, "clip_ratio/region_mean": 0.023602438159286976, "reward_total_mean": 0.6469426155090332, "reward_meter_mean": 0.6469426155090332, "reward_meter_std": 0.24868278205394745, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6469426155090332, "reward_total_composite_std": 0.24868278205394745} {"timestamp_utc": "2026-04-11T22:55:49Z", "mode": "train", "global_step": 789, "epoch": 0.03169056512832871, "loss": 0.0131, "grad_norm": 13.922452926635742, "learning_rate": 7.612121212121213e-06, "num_tokens": 1743054.0, "completions/mean_length": 31.25, "completions/min_length": 30.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.25, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9813005924224854, "rewards/meter/std": 0.016189413145184517, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9813005924224854, "rewards/total_composite/std": 0.016189413145184517, "reward": 0.9813005924224854, "reward_std": 0.016189415007829666, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.033597491681575775, "sampling/sampling_logp_difference/max": 1.0770900249481201, "sampling/importance_sampling_ratio/min": 0.3405851721763611, "sampling/importance_sampling_ratio/mean": 0.9976152181625366, "sampling/importance_sampling_ratio/max": 1.8372166156768799, "entropy": 0.17746192403137684, "clip_ratio/low_mean": 0.008072916883975267, "clip_ratio/low_min": 0.008072916883975267, "clip_ratio/high_mean": 0.008072916883975267, "clip_ratio/high_max": 0.008072916883975267, "clip_ratio/region_mean": 0.016145833767950535, "reward_total_mean": 0.9813005924224854, "reward_meter_mean": 0.9813005924224854, "reward_meter_std": 0.016189413145184517, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9813005924224854, "reward_total_composite_std": 0.016189413145184517} {"timestamp_utc": "2026-04-11T22:55:54Z", "mode": "train", "global_step": 790, "epoch": 0.03173073061011367, "loss": -0.0039, "grad_norm": 6.977390766143799, "learning_rate": 7.609090909090909e-06, "num_tokens": 1745187.0, "completions/mean_length": 89.625, "completions/min_length": 85.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.625, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.8486406207084656, "rewards/meter/std": 0.2562169134616852, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6789125204086304, "rewards/total_composite/std": 0.20497353374958038, "reward": 0.6789125204086304, "reward_std": 0.20497353374958038, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021279064938426018, "sampling/sampling_logp_difference/max": 0.814723014831543, "sampling/importance_sampling_ratio/min": 0.44276192784309387, "sampling/importance_sampling_ratio/mean": 1.0061084032058716, "sampling/importance_sampling_ratio/max": 1.8353419303894043, "entropy": 0.11805877834558487, "clip_ratio/low_mean": 0.009868005756288767, "clip_ratio/low_min": 0.009868005756288767, "clip_ratio/high_mean": 0.019598963437601924, "clip_ratio/high_max": 0.019598963437601924, "clip_ratio/region_mean": 0.02946696919389069, "reward_total_mean": 0.6789125204086304, "reward_meter_mean": 0.8486406207084656, "reward_meter_std": 0.2562169134616852, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6789125204086304, "reward_total_composite_std": 0.20497353374958038} {"timestamp_utc": "2026-04-11T22:56:01Z", "mode": "train", "global_step": 791, "epoch": 0.03177089609189862, "loss": -0.0395, "grad_norm": 1.8626375198364258, "learning_rate": 7.606060606060606e-06, "num_tokens": 1748773.0, "completions/mean_length": 247.25, "completions/min_length": 228.0, "completions/max_length": 258.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 247.25, "completions/min_terminated_length": 228.0, "completions/max_terminated_length": 258.0, "rewards/meter/mean": 0.9959322214126587, "rewards/meter/std": 0.004204806871712208, "rewards/count_adherence/mean": 0.910714328289032, "rewards/count_adherence/std": 0.07393559068441391, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.49047619104385376, "rewards/repeat_penalty/std": 0.17105023562908173, "rewards/total_composite/mean": 0.4338645935058594, "rewards/total_composite/std": 0.12500374019145966, "reward": 0.4338645935058594, "reward_std": 0.12500372529029846, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014479896053671837, "sampling/sampling_logp_difference/max": 2.833479881286621, "sampling/importance_sampling_ratio/min": 0.0588078573346138, "sampling/importance_sampling_ratio/mean": 1.0004253387451172, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.042171002831310034, "clip_ratio/low_mean": 0.0032260402804240584, "clip_ratio/low_min": 0.0032260402804240584, "clip_ratio/high_mean": 0.00779381615575403, "clip_ratio/high_max": 0.00779381615575403, "clip_ratio/region_mean": 0.011019856436178088, "reward_total_mean": 0.4338645935058594, "reward_meter_mean": 0.9959322214126587, "reward_meter_std": 0.004204806871712208, "reward_count_adherence_mean": 0.910714328289032, "reward_count_adherence_std": 0.07393559068441391, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.49047619104385376, "reward_repeat_penalty_std": 0.17105023562908173, "reward_total_composite_mean": 0.4338645935058594, "reward_total_composite_std": 0.12500374019145966} {"timestamp_utc": "2026-04-11T22:56:06Z", "mode": "train", "global_step": 792, "epoch": 0.031811061573683574, "loss": 0.023, "grad_norm": 3.4738333225250244, "learning_rate": 7.603030303030303e-06, "num_tokens": 1751485.0, "completions/mean_length": 135.0, "completions/min_length": 124.0, "completions/max_length": 148.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 135.0, "completions/min_terminated_length": 124.0, "completions/max_terminated_length": 148.0, "rewards/meter/mean": 0.991753101348877, "rewards/meter/std": 0.002629074966534972, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6611686944961548, "rewards/total_composite/std": 0.0017527303425595164, "reward": 0.6611686944961548, "reward_std": 0.0017527244053781033, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019971484318375587, "sampling/sampling_logp_difference/max": 2.0009312629699707, "sampling/importance_sampling_ratio/min": 0.13520930707454681, "sampling/importance_sampling_ratio/mean": 1.0068343877792358, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0634374669753015, "clip_ratio/low_mean": 0.010955466306768358, "clip_ratio/low_min": 0.010955466306768358, "clip_ratio/high_mean": 0.00658258656039834, "clip_ratio/high_max": 0.00658258656039834, "clip_ratio/region_mean": 0.017538052867166698, "reward_total_mean": 0.6611686944961548, "reward_meter_mean": 0.991753101348877, "reward_meter_std": 0.002629074966534972, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6611686944961548, "reward_total_composite_std": 0.0017527303425595164} {"timestamp_utc": "2026-04-11T22:56:11Z", "mode": "train", "global_step": 793, "epoch": 0.03185122705546853, "loss": 0.0066, "grad_norm": 5.889811038970947, "learning_rate": 7.600000000000001e-06, "num_tokens": 1753253.0, "completions/mean_length": 56.0, "completions/min_length": 51.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.0, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9871419668197632, "rewards/meter/std": 0.005153529345989227, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.9050641655921936, "rewards/total_composite/std": 0.15341757237911224, "reward": 0.9050641655921936, "reward_std": 0.15341757237911224, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.054076679050922394, "sampling/sampling_logp_difference/max": 2.979457139968872, "sampling/importance_sampling_ratio/min": 0.05082041397690773, "sampling/importance_sampling_ratio/mean": 0.9988625645637512, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15502143744379282, "clip_ratio/low_mean": 0.006658692145720124, "clip_ratio/low_min": 0.006658692145720124, "clip_ratio/high_mean": 0.0335467669647187, "clip_ratio/high_max": 0.0335467669647187, "clip_ratio/region_mean": 0.040205459110438824, "reward_total_mean": 0.9050641655921936, "reward_meter_mean": 0.9871419668197632, "reward_meter_std": 0.005153529345989227, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_total_composite_mean": 0.9050641655921936, "reward_total_composite_std": 0.15341757237911224} {"timestamp_utc": "2026-04-11T22:56:16Z", "mode": "train", "global_step": 794, "epoch": 0.03189139253725348, "loss": 0.002, "grad_norm": 6.565411567687988, "learning_rate": 7.596969696969697e-06, "num_tokens": 1755042.0, "completions/mean_length": 71.625, "completions/min_length": 69.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.625, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9123142957687378, "rewards/meter/std": 0.09913700073957443, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9123142957687378, "rewards/total_composite/std": 0.09913700073957443, "reward": 0.9123142957687378, "reward_std": 0.09913701564073563, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.034922052174806595, "sampling/sampling_logp_difference/max": 1.3885717391967773, "sampling/importance_sampling_ratio/min": 0.249431312084198, "sampling/importance_sampling_ratio/mean": 1.0043187141418457, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15924948174506426, "clip_ratio/low_mean": 0.019052310031838715, "clip_ratio/low_min": 0.019052310031838715, "clip_ratio/high_mean": 0.007044379832223058, "clip_ratio/high_max": 0.007044379832223058, "clip_ratio/region_mean": 0.026096689864061773, "reward_total_mean": 0.9123142957687378, "reward_meter_mean": 0.9123142957687378, "reward_meter_std": 0.09913700073957443, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9123142957687378, "reward_total_composite_std": 0.09913700073957443} {"timestamp_utc": "2026-04-11T22:56:21Z", "mode": "train", "global_step": 795, "epoch": 0.031931558019038436, "loss": 0.0046, "grad_norm": 2.6046133041381836, "learning_rate": 7.593939393939395e-06, "num_tokens": 1756866.0, "completions/mean_length": 73.0, "completions/min_length": 73.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.0, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.613559365272522, "rewards/meter/std": 0.046057041734457016, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.613559365272522, "rewards/total_composite/std": 0.046057041734457016, "reward": 0.613559365272522, "reward_std": 0.04605703800916672, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01630816049873829, "sampling/sampling_logp_difference/max": 0.9375072717666626, "sampling/importance_sampling_ratio/min": 0.5393000841140747, "sampling/importance_sampling_ratio/mean": 1.0046606063842773, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07319271843880415, "clip_ratio/low_mean": 0.008561643771827221, "clip_ratio/low_min": 0.008561643771827221, "clip_ratio/high_mean": 0.0034246575087308884, "clip_ratio/high_max": 0.0034246575087308884, "clip_ratio/region_mean": 0.01198630128055811, "reward_total_mean": 0.613559365272522, "reward_meter_mean": 0.613559365272522, "reward_meter_std": 0.046057041734457016, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.613559365272522, "reward_total_composite_std": 0.046057041734457016} {"timestamp_utc": "2026-04-11T22:56:28Z", "mode": "train", "global_step": 796, "epoch": 0.03197172350082339, "loss": 0.0182, "grad_norm": 1.5290881395339966, "learning_rate": 7.590909090909091e-06, "num_tokens": 1760371.0, "completions/mean_length": 250.125, "completions/min_length": 248.0, "completions/max_length": 264.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 250.125, "completions/min_terminated_length": 248.0, "completions/max_terminated_length": 264.0, "rewards/meter/mean": 0.6284428834915161, "rewards/meter/std": 0.1288241744041443, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6098901033401489, "rewards/repeat_penalty/std": 0.015540807507932186, "rewards/total_composite/mean": 0.3849101662635803, "rewards/total_composite/std": 0.08409524708986282, "reward": 0.3849101662635803, "reward_std": 0.08409524708986282, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006375753786414862, "sampling/sampling_logp_difference/max": 1.0706424713134766, "sampling/importance_sampling_ratio/min": 0.3427882194519043, "sampling/importance_sampling_ratio/mean": 1.0002148151397705, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.023278776556253433, "clip_ratio/low_mean": 0.0004734848625957966, "clip_ratio/low_min": 0.0004734848625957966, "clip_ratio/high_mean": 0.002516112755984068, "clip_ratio/high_max": 0.002516112755984068, "clip_ratio/region_mean": 0.0029895976185798645, "reward_total_mean": 0.3849101662635803, "reward_meter_mean": 0.6284428834915161, "reward_meter_std": 0.1288241744041443, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6098901033401489, "reward_repeat_penalty_std": 0.015540807507932186, "reward_total_composite_mean": 0.3849101662635803, "reward_total_composite_std": 0.08409524708986282} {"timestamp_utc": "2026-04-11T22:56:33Z", "mode": "train", "global_step": 797, "epoch": 0.032011888982608344, "loss": -0.0121, "grad_norm": 8.541836738586426, "learning_rate": 7.587878787878788e-06, "num_tokens": 1762455.0, "completions/mean_length": 77.5, "completions/min_length": 72.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.5, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.9952297210693359, "rewards/meter/std": 0.002311403863132, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.891382098197937, "rewards/total_composite/std": 0.1964196413755417, "reward": 0.891382098197937, "reward_std": 0.1964196413755417, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.034574542194604874, "sampling/sampling_logp_difference/max": 3.3448140621185303, "sampling/importance_sampling_ratio/min": 0.03526677191257477, "sampling/importance_sampling_ratio/mean": 0.9954510927200317, "sampling/importance_sampling_ratio/max": 1.482370376586914, "entropy": 0.14471599273383617, "clip_ratio/low_mean": 0.003246753243729472, "clip_ratio/low_min": 0.003246753243729472, "clip_ratio/high_mean": 0.009473072132095695, "clip_ratio/high_max": 0.009473072132095695, "clip_ratio/region_mean": 0.012719825375825167, "reward_total_mean": 0.891382098197937, "reward_meter_mean": 0.9952297210693359, "reward_meter_std": 0.002311403863132, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.891382098197937, "reward_total_composite_std": 0.1964196413755417} {"timestamp_utc": "2026-04-11T22:56:38Z", "mode": "train", "global_step": 798, "epoch": 0.0320520544643933, "loss": 0.0167, "grad_norm": 6.7140374183654785, "learning_rate": 7.584848484848486e-06, "num_tokens": 1764291.0, "completions/mean_length": 66.5, "completions/min_length": 64.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.5, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.6813795566558838, "rewards/meter/std": 0.04148668423295021, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6813795566558838, "rewards/total_composite/std": 0.04148668423295021, "reward": 0.6813795566558838, "reward_std": 0.04148669168353081, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023791270330548286, "sampling/sampling_logp_difference/max": 0.9566974639892578, "sampling/importance_sampling_ratio/min": 0.5285674333572388, "sampling/importance_sampling_ratio/mean": 1.010377049446106, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1323620891198516, "clip_ratio/low_mean": 0.020607191254384816, "clip_ratio/low_min": 0.020607191254384816, "clip_ratio/high_mean": 0.00390625, "clip_ratio/high_max": 0.00390625, "clip_ratio/region_mean": 0.024513441254384816, "reward_total_mean": 0.6813795566558838, "reward_meter_mean": 0.6813795566558838, "reward_meter_std": 0.04148668423295021, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6813795566558838, "reward_total_composite_std": 0.04148668423295021} {"timestamp_utc": "2026-04-11T22:56:45Z", "mode": "train", "global_step": 799, "epoch": 0.03209221994617825, "loss": 0.0044, "grad_norm": 1.2270196676254272, "learning_rate": 7.581818181818183e-06, "num_tokens": 1767790.0, "completions/mean_length": 239.375, "completions/min_length": 237.0, "completions/max_length": 240.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 239.375, "completions/min_terminated_length": 237.0, "completions/max_terminated_length": 240.0, "rewards/meter/mean": 0.7462765574455261, "rewards/meter/std": 0.09833642095327377, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6000000238418579, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.38379937410354614, "rewards/total_composite/std": 0.05057300627231598, "reward": 0.38379937410354614, "reward_std": 0.050573013722896576, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007426128257066011, "sampling/sampling_logp_difference/max": 2.6020090579986572, "sampling/importance_sampling_ratio/min": 0.07412450760602951, "sampling/importance_sampling_ratio/mean": 0.9996190071105957, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.030161422211676836, "clip_ratio/low_mean": 0.003645833523478359, "clip_ratio/low_min": 0.003645833523478359, "clip_ratio/high_mean": 0.002109704539179802, "clip_ratio/high_max": 0.002109704539179802, "clip_ratio/region_mean": 0.005755538062658161, "reward_total_mean": 0.38379937410354614, "reward_meter_mean": 0.7462765574455261, "reward_meter_std": 0.09833642095327377, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6000000238418579, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.38379937410354614, "reward_total_composite_std": 0.05057300627231598} {"timestamp_utc": "2026-04-11T22:56:50Z", "mode": "train", "global_step": 800, "epoch": 0.032132385427963206, "loss": 0.0432, "grad_norm": 4.79632568359375, "learning_rate": 7.57878787878788e-06, "num_tokens": 1770025.0, "completions/mean_length": 119.375, "completions/min_length": 110.0, "completions/max_length": 130.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 119.375, "completions/min_terminated_length": 110.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.5170607566833496, "rewards/meter/std": 0.48065322637557983, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.4120614528656006, "rewards/total_composite/std": 0.386055052280426, "reward": 0.4120614528656006, "reward_std": 0.386055052280426, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04103344678878784, "sampling/sampling_logp_difference/max": 1.1491317749023438, "sampling/importance_sampling_ratio/min": 0.3169117867946625, "sampling/importance_sampling_ratio/mean": 0.9977750182151794, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23464529775083065, "clip_ratio/low_mean": 0.024260351667180657, "clip_ratio/low_min": 0.024260351667180657, "clip_ratio/high_mean": 0.0143592240056023, "clip_ratio/high_max": 0.0143592240056023, "clip_ratio/region_mean": 0.03861957567278296, "reward_total_mean": 0.4120614528656006, "reward_meter_mean": 0.5170607566833496, "reward_meter_std": 0.48065322637557983, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.10690449178218842, "reward_total_composite_mean": 0.4120614528656006, "reward_total_composite_std": 0.386055052280426} {"timestamp_utc": "2026-04-11T22:58:16Z", "mode": "eval", "global_step": 800, "epoch": 0.032132385427963206, "eval_loss": NaN, "eval_runtime": 85.9157, "eval_samples_per_second": 1.21, "eval_steps_per_second": 0.151, "eval_num_tokens": 1770025.0, "eval_completions/mean_length": 234.29807692307693, "eval_completions/min_length": 63.84615384615385, "eval_completions/max_length": 461.9230769230769, "eval_completions/clipped_ratio": 0.057692307692307696, "eval_completions/mean_terminated_length": 216.9835216815655, "eval_completions/min_terminated_length": 63.84615384615385, "eval_completions/max_terminated_length": 414.46153846153845, "eval_rewards/meter/mean": 0.6419616112342248, "eval_rewards/meter/std": 0.3754527878302794, "eval_rewards/count_adherence/mean": 0.9552615697567279, "eval_rewards/count_adherence/std": 0.08013723160211857, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/repeat_penalty/mean": 0.6556147245260385, "eval_rewards/repeat_penalty/std": 0.24277657327743676, "eval_rewards/total_composite/mean": 0.3991968219096844, "eval_rewards/total_composite/std": 0.3098094039238416, "eval_reward": 0.3991968219096844, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.0054089168552309275, "eval_sampling/sampling_logp_difference/max": 0.7075341939926147, "eval_sampling/importance_sampling_ratio/min": 0.5230464408030877, "eval_sampling/importance_sampling_ratio/mean": 1.0009051194557776, "eval_sampling/importance_sampling_ratio/max": 1.2961730773632343, "eval_entropy": 0.047992275741237864, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.3991968219096844, "eval_reward_meter_mean": 0.6419616112342248, "eval_reward_meter_std": 0.3754527878302794, "eval_reward_count_adherence_mean": 0.9552615697567279, "eval_reward_count_adherence_std": 0.08013723160211857, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_repeat_penalty_mean": 0.6556147245260385, "eval_reward_repeat_penalty_std": 0.24277657327743676, "eval_reward_total_composite_mean": 0.3991968219096844, "eval_reward_total_composite_std": 0.3098094039238416} {"timestamp_utc": "2026-04-11T22:58:23Z", "mode": "train", "global_step": 801, "epoch": 0.03217255090974816, "loss": 0.0018, "grad_norm": 4.925754070281982, "learning_rate": 7.5757575757575764e-06, "num_tokens": 1771810.0, "completions/mean_length": 67.125, "completions/min_length": 67.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.125, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.7430859804153442, "rewards/meter/std": 0.04014641046524048, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7430859804153442, "rewards/total_composite/std": 0.04014641046524048, "reward": 0.7430859804153442, "reward_std": 0.040146395564079285, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02094501443207264, "sampling/sampling_logp_difference/max": 1.098017692565918, "sampling/importance_sampling_ratio/min": 0.33353158831596375, "sampling/importance_sampling_ratio/mean": 1.0070921182632446, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08063888642936945, "clip_ratio/low_mean": 0.018601843621581793, "clip_ratio/low_min": 0.018601843621581793, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.018601843621581793, "reward_total_mean": 0.7430859804153442, "reward_meter_mean": 0.7430859804153442, "reward_meter_std": 0.04014641046524048, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7430859804153442, "reward_total_composite_std": 0.04014641046524048} {"timestamp_utc": "2026-04-11T22:58:28Z", "mode": "train", "global_step": 802, "epoch": 0.032212716391533114, "loss": 0.0172, "grad_norm": 7.339356422424316, "learning_rate": 7.572727272727274e-06, "num_tokens": 1773359.0, "completions/mean_length": 37.625, "completions/min_length": 36.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.625, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.13143374025821686, "rewards/meter/std": 0.11004206538200378, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.13143374025821686, "rewards/total_composite/std": 0.11004206538200378, "reward": 0.13143374025821686, "reward_std": 0.11004206538200378, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03102065436542034, "sampling/sampling_logp_difference/max": 1.9875779151916504, "sampling/importance_sampling_ratio/min": 0.13702690601348877, "sampling/importance_sampling_ratio/mean": 0.9923664331436157, "sampling/importance_sampling_ratio/max": 1.671715497970581, "entropy": 0.1240494973026216, "clip_ratio/low_mean": 0.009703947464004159, "clip_ratio/low_min": 0.009703947464004159, "clip_ratio/high_mean": 0.00995732587762177, "clip_ratio/high_max": 0.00995732587762177, "clip_ratio/region_mean": 0.01966127334162593, "reward_total_mean": 0.13143374025821686, "reward_meter_mean": 0.13143374025821686, "reward_meter_std": 0.11004206538200378, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.13143374025821686, "reward_total_composite_std": 0.11004206538200378} {"timestamp_utc": "2026-04-11T22:58:32Z", "mode": "train", "global_step": 803, "epoch": 0.03225288187331807, "loss": 0.0049, "grad_norm": 4.46676778793335, "learning_rate": 7.56969696969697e-06, "num_tokens": 1775325.0, "completions/mean_length": 70.75, "completions/min_length": 70.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.75, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9594069123268127, "rewards/meter/std": 0.05366376414895058, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9594069123268127, "rewards/total_composite/std": 0.05366376414895058, "reward": 0.9594069123268127, "reward_std": 0.05366375669836998, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03665493428707123, "sampling/sampling_logp_difference/max": 1.9053394794464111, "sampling/importance_sampling_ratio/min": 0.14877213537693024, "sampling/importance_sampling_ratio/mean": 0.9985708594322205, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1510819010436535, "clip_ratio/low_mean": 0.005281690042465925, "clip_ratio/low_min": 0.005281690042465925, "clip_ratio/high_mean": 0.026509752846322954, "clip_ratio/high_max": 0.026509752846322954, "clip_ratio/region_mean": 0.03179144288878888, "reward_total_mean": 0.9594069123268127, "reward_meter_mean": 0.9594069123268127, "reward_meter_std": 0.05366376414895058, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9594069123268127, "reward_total_composite_std": 0.05366376414895058} {"timestamp_utc": "2026-04-11T22:58:44Z", "mode": "train", "global_step": 804, "epoch": 0.03229304735510302, "loss": 0.0506, "grad_norm": 0.5559709072113037, "learning_rate": 7.566666666666667e-06, "num_tokens": 1780007.0, "completions/mean_length": 505.25, "completions/min_length": 493.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 503.0, "completions/min_terminated_length": 493.0, "completions/max_terminated_length": 508.0, "rewards/meter/mean": 0.9569629430770874, "rewards/meter/std": 0.102637879550457, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.0235702246427536, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4376780688762665, "rewards/repeat_penalty/std": 0.22492921352386475, "rewards/total_composite/mean": 0.38479870557785034, "rewards/total_composite/std": 0.20677605271339417, "reward": 0.38479870557785034, "reward_std": 0.20677603781223297, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006570629775524139, "sampling/sampling_logp_difference/max": 2.7531256675720215, "sampling/importance_sampling_ratio/min": 0.06372835487127304, "sampling/importance_sampling_ratio/mean": 1.0004603862762451, "sampling/importance_sampling_ratio/max": 1.9659916162490845, "entropy": 0.014583299867808819, "clip_ratio/low_mean": 0.0004926113178953528, "clip_ratio/low_min": 0.0004926113178953528, "clip_ratio/high_mean": 0.003243225917685777, "clip_ratio/high_max": 0.003243225917685777, "clip_ratio/region_mean": 0.00373583723558113, "reward_total_mean": 0.38479870557785034, "reward_meter_mean": 0.9569629430770874, "reward_meter_std": 0.102637879550457, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.0235702246427536, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4376780688762665, "reward_repeat_penalty_std": 0.22492921352386475, "reward_total_composite_mean": 0.38479870557785034, "reward_total_composite_std": 0.20677605271339417} {"timestamp_utc": "2026-04-11T22:58:48Z", "mode": "train", "global_step": 805, "epoch": 0.032333212836887976, "loss": 0.0024, "grad_norm": 4.009762287139893, "learning_rate": 7.563636363636364e-06, "num_tokens": 1781817.0, "completions/mean_length": 67.25, "completions/min_length": 67.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.25, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.7574520111083984, "rewards/meter/std": 0.03009505569934845, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7574520111083984, "rewards/total_composite/std": 0.03009505569934845, "reward": 0.7574520111083984, "reward_std": 0.0300950538367033, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020206134766340256, "sampling/sampling_logp_difference/max": 1.227433443069458, "sampling/importance_sampling_ratio/min": 0.29304373264312744, "sampling/importance_sampling_ratio/mean": 0.9979568719863892, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.05657489877194166, "clip_ratio/low_mean": 0.011085874866694212, "clip_ratio/low_min": 0.011085874866694212, "clip_ratio/high_mean": 0.00932835799176246, "clip_ratio/high_max": 0.00932835799176246, "clip_ratio/region_mean": 0.02041423285845667, "reward_total_mean": 0.7574520111083984, "reward_meter_mean": 0.7574520111083984, "reward_meter_std": 0.03009505569934845, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7574520111083984, "reward_total_composite_std": 0.03009505569934845} {"timestamp_utc": "2026-04-11T22:58:53Z", "mode": "train", "global_step": 806, "epoch": 0.03237337831867293, "loss": 0.0021, "grad_norm": 3.6810977458953857, "learning_rate": 7.560606060606062e-06, "num_tokens": 1783594.0, "completions/mean_length": 58.125, "completions/min_length": 57.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.125, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.990125298500061, "rewards/meter/std": 0.0014512698398903012, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8333333730697632, "rewards/repeat_penalty/std": 0.17817415297031403, "rewards/total_composite/mean": 0.8252000212669373, "rewards/total_composite/std": 0.17693018913269043, "reward": 0.8252000212669373, "reward_std": 0.17693018913269043, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023440338671207428, "sampling/sampling_logp_difference/max": 1.1792564392089844, "sampling/importance_sampling_ratio/min": 0.30750730633735657, "sampling/importance_sampling_ratio/mean": 0.9979320168495178, "sampling/importance_sampling_ratio/max": 1.3977869749069214, "entropy": 0.11172830406576395, "clip_ratio/low_mean": 0.004310344811528921, "clip_ratio/low_min": 0.004310344811528921, "clip_ratio/high_mean": 0.01062974869273603, "clip_ratio/high_max": 0.01062974869273603, "clip_ratio/region_mean": 0.01494009350426495, "reward_total_mean": 0.8252000212669373, "reward_meter_mean": 0.990125298500061, "reward_meter_std": 0.0014512698398903012, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8333333730697632, "reward_repeat_penalty_std": 0.17817415297031403, "reward_total_composite_mean": 0.8252000212669373, "reward_total_composite_std": 0.17693018913269043} {"timestamp_utc": "2026-04-11T22:58:58Z", "mode": "train", "global_step": 807, "epoch": 0.032413543800457884, "loss": -0.0084, "grad_norm": 5.640214920043945, "learning_rate": 7.557575757575758e-06, "num_tokens": 1785577.0, "completions/mean_length": 75.875, "completions/min_length": 72.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.875, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.7918260097503662, "rewards/meter/std": 0.3541868031024933, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7303093671798706, "rewards/total_composite/std": 0.35869720578193665, "reward": 0.7303093671798706, "reward_std": 0.35869717597961426, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02571277692914009, "sampling/sampling_logp_difference/max": 1.1890771389007568, "sampling/importance_sampling_ratio/min": 0.4486599266529083, "sampling/importance_sampling_ratio/mean": 1.0031059980392456, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12689008563756943, "clip_ratio/low_mean": 0.00854225957300514, "clip_ratio/low_min": 0.00854225957300514, "clip_ratio/high_mean": 0.008098726975731552, "clip_ratio/high_max": 0.008098726975731552, "clip_ratio/region_mean": 0.01664098654873669, "reward_total_mean": 0.7303093671798706, "reward_meter_mean": 0.7918260097503662, "reward_meter_std": 0.3541868031024933, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7303093671798706, "reward_total_composite_std": 0.35869720578193665} {"timestamp_utc": "2026-04-11T22:59:02Z", "mode": "train", "global_step": 808, "epoch": 0.03245370928224284, "loss": -0.0018, "grad_norm": 14.179214477539062, "learning_rate": 7.5545454545454555e-06, "num_tokens": 1787137.0, "completions/mean_length": 38.0, "completions/min_length": 38.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.0, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.8494601845741272, "rewards/meter/std": 0.3245563507080078, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8494601845741272, "rewards/total_composite/std": 0.3245563507080078, "reward": 0.8494601845741272, "reward_std": 0.3245563209056854, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026692412793636322, "sampling/sampling_logp_difference/max": 2.183236598968506, "sampling/importance_sampling_ratio/min": 0.11267625540494919, "sampling/importance_sampling_ratio/mean": 0.9977914690971375, "sampling/importance_sampling_ratio/max": 1.5366783142089844, "entropy": 0.09128655772656202, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.016447368543595076, "clip_ratio/high_max": 0.016447368543595076, "clip_ratio/region_mean": 0.016447368543595076, "reward_total_mean": 0.8494601845741272, "reward_meter_mean": 0.8494601845741272, "reward_meter_std": 0.3245563507080078, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8494601845741272, "reward_total_composite_std": 0.3245563507080078} {"timestamp_utc": "2026-04-11T22:59:09Z", "mode": "train", "global_step": 809, "epoch": 0.03249387476402779, "loss": -0.0031, "grad_norm": 1.1023591756820679, "learning_rate": 7.551515151515152e-06, "num_tokens": 1790539.0, "completions/mean_length": 218.25, "completions/min_length": 206.0, "completions/max_length": 230.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 218.25, "completions/min_terminated_length": 206.0, "completions/max_terminated_length": 230.0, "rewards/meter/mean": 0.9922986030578613, "rewards/meter/std": 0.004149852320551872, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5681818723678589, "rewards/repeat_penalty/std": 0.1735115498304367, "rewards/total_composite/mean": 0.5638895034790039, "rewards/total_composite/std": 0.17219732701778412, "reward": 0.5638895034790039, "reward_std": 0.17219732701778412, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011151120997965336, "sampling/sampling_logp_difference/max": 1.5731306076049805, "sampling/importance_sampling_ratio/min": 0.20739488303661346, "sampling/importance_sampling_ratio/mean": 1.0026495456695557, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03606141824275255, "clip_ratio/low_mean": 0.005849295761436224, "clip_ratio/low_min": 0.005849295761436224, "clip_ratio/high_mean": 0.003968499368056655, "clip_ratio/high_max": 0.003968499368056655, "clip_ratio/region_mean": 0.009817795129492879, "reward_total_mean": 0.5638895034790039, "reward_meter_mean": 0.9922986030578613, "reward_meter_std": 0.004149852320551872, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5681818723678589, "reward_repeat_penalty_std": 0.1735115498304367, "reward_total_composite_mean": 0.5638895034790039, "reward_total_composite_std": 0.17219732701778412} {"timestamp_utc": "2026-04-11T22:59:14Z", "mode": "train", "global_step": 810, "epoch": 0.032534040245812745, "loss": 0.0017, "grad_norm": 5.80548095703125, "learning_rate": 7.548484848484849e-06, "num_tokens": 1792366.0, "completions/mean_length": 71.375, "completions/min_length": 70.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.375, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9742876887321472, "rewards/meter/std": 0.029914017766714096, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9332020878791809, "rewards/total_composite/std": 0.11529982835054398, "reward": 0.9332020878791809, "reward_std": 0.11529984325170517, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0373886339366436, "sampling/sampling_logp_difference/max": 1.9366556406021118, "sampling/importance_sampling_ratio/min": 0.14418534934520721, "sampling/importance_sampling_ratio/mean": 1.0060287714004517, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15044390503317118, "clip_ratio/low_mean": 0.005306840990670025, "clip_ratio/low_min": 0.005306840990670025, "clip_ratio/high_mean": 0.03143186215311289, "clip_ratio/high_max": 0.03143186215311289, "clip_ratio/region_mean": 0.036738703143782914, "reward_total_mean": 0.9332020878791809, "reward_meter_mean": 0.9742876887321472, "reward_meter_std": 0.029914017766714096, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.9332020878791809, "reward_total_composite_std": 0.11529982835054398} {"timestamp_utc": "2026-04-11T22:59:20Z", "mode": "train", "global_step": 811, "epoch": 0.0325742057275977, "loss": 0.0192, "grad_norm": 2.172598123550415, "learning_rate": 7.545454545454546e-06, "num_tokens": 1795876.0, "completions/mean_length": 203.75, "completions/min_length": 196.0, "completions/max_length": 229.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 203.75, "completions/min_terminated_length": 196.0, "completions/max_terminated_length": 229.0, "rewards/meter/mean": 0.710443377494812, "rewards/meter/std": 0.2666179835796356, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4431818127632141, "rewards/repeat_penalty/std": 0.22498852014541626, "rewards/total_composite/mean": 0.3340369462966919, "rewards/total_composite/std": 0.2214316725730896, "reward": 0.3340369462966919, "reward_std": 0.2214316576719284, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021875398233532906, "sampling/sampling_logp_difference/max": 4.384680271148682, "sampling/importance_sampling_ratio/min": 0.012466873973608017, "sampling/importance_sampling_ratio/mean": 1.0012760162353516, "sampling/importance_sampling_ratio/max": 1.9448559284210205, "entropy": 0.1404771413654089, "clip_ratio/low_mean": 0.006675369921140373, "clip_ratio/low_min": 0.006675369921140373, "clip_ratio/high_mean": 0.011296228156425059, "clip_ratio/high_max": 0.011296228156425059, "clip_ratio/region_mean": 0.01797159807756543, "reward_total_mean": 0.3340369462966919, "reward_meter_mean": 0.710443377494812, "reward_meter_std": 0.2666179835796356, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4431818127632141, "reward_repeat_penalty_std": 0.22498852014541626, "reward_total_composite_mean": 0.3340369462966919, "reward_total_composite_std": 0.2214316725730896} {"timestamp_utc": "2026-04-11T22:59:25Z", "mode": "train", "global_step": 812, "epoch": 0.03261437120938265, "loss": 0.0139, "grad_norm": 8.286386489868164, "learning_rate": 7.542424242424244e-06, "num_tokens": 1797566.0, "completions/mean_length": 65.25, "completions/min_length": 65.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.25, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9978810548782349, "rewards/meter/std": 0.0005103643052279949, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9563552141189575, "rewards/total_composite/std": 0.11796265840530396, "reward": 0.9563552141189575, "reward_std": 0.11796264350414276, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007206229493021965, "sampling/sampling_logp_difference/max": 0.5880947113037109, "sampling/importance_sampling_ratio/min": 0.5553844571113586, "sampling/importance_sampling_ratio/mean": 1.002901315689087, "sampling/importance_sampling_ratio/max": 1.3499990701675415, "entropy": 0.036765412194654346, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.005769230774603784, "clip_ratio/high_max": 0.005769230774603784, "clip_ratio/region_mean": 0.005769230774603784, "reward_total_mean": 0.9563552141189575, "reward_meter_mean": 0.9978810548782349, "reward_meter_std": 0.0005103643052279949, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.9563552141189575, "reward_total_composite_std": 0.11796265840530396} {"timestamp_utc": "2026-04-11T22:59:34Z", "mode": "train", "global_step": 813, "epoch": 0.03265453669116761, "loss": 0.0079, "grad_norm": 1.0310794115066528, "learning_rate": 7.53939393939394e-06, "num_tokens": 1801761.0, "completions/mean_length": 332.375, "completions/min_length": 302.0, "completions/max_length": 353.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 332.375, "completions/min_terminated_length": 302.0, "completions/max_terminated_length": 353.0, "rewards/meter/mean": 0.9938352704048157, "rewards/meter/std": 0.008550022728741169, "rewards/count_adherence/mean": 0.890625, "rewards/count_adherence/std": 0.04419417306780815, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5850183963775635, "rewards/repeat_penalty/std": 0.009098809212446213, "rewards/total_composite/mean": 0.517345666885376, "rewards/total_composite/std": 0.012124452739953995, "reward": 0.517345666885376, "reward_std": 0.012124458327889442, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0052232746966183186, "sampling/sampling_logp_difference/max": 0.7512289881706238, "sampling/importance_sampling_ratio/min": 0.47178637981414795, "sampling/importance_sampling_ratio/mean": 1.0009092092514038, "sampling/importance_sampling_ratio/max": 1.6740483045578003, "entropy": 0.02348946128040552, "clip_ratio/low_mean": 0.005218350415816531, "clip_ratio/low_min": 0.005218350415816531, "clip_ratio/high_mean": 0.0019113150192424655, "clip_ratio/high_max": 0.0019113150192424655, "clip_ratio/region_mean": 0.007129665435058996, "reward_total_mean": 0.517345666885376, "reward_meter_mean": 0.9938352704048157, "reward_meter_std": 0.008550022728741169, "reward_count_adherence_mean": 0.890625, "reward_count_adherence_std": 0.04419417306780815, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5850183963775635, "reward_repeat_penalty_std": 0.009098809212446213, "reward_total_composite_mean": 0.517345666885376, "reward_total_composite_std": 0.012124452739953995} {"timestamp_utc": "2026-04-11T22:59:40Z", "mode": "train", "global_step": 814, "epoch": 0.03269470217295257, "loss": -0.001, "grad_norm": 13.524036407470703, "learning_rate": 7.536363636363637e-06, "num_tokens": 1804516.0, "completions/mean_length": 159.375, "completions/min_length": 158.0, "completions/max_length": 167.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 159.375, "completions/min_terminated_length": 158.0, "completions/max_terminated_length": 167.0, "rewards/meter/mean": 0.9979825019836426, "rewards/meter/std": 0.0006693408940918744, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6653216481208801, "rewards/total_composite/std": 0.00044622019049711525, "reward": 0.6653216481208801, "reward_std": 0.00044622053974308074, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005735776387155056, "sampling/sampling_logp_difference/max": 2.232571601867676, "sampling/importance_sampling_ratio/min": 0.10725226998329163, "sampling/importance_sampling_ratio/mean": 0.9996488094329834, "sampling/importance_sampling_ratio/max": 1.493585467338562, "entropy": 0.03128327080048621, "clip_ratio/low_mean": 0.00237341778120026, "clip_ratio/low_min": 0.00237341778120026, "clip_ratio/high_mean": 0.00237341778120026, "clip_ratio/high_max": 0.00237341778120026, "clip_ratio/region_mean": 0.00474683556240052, "reward_total_mean": 0.6653216481208801, "reward_meter_mean": 0.9979825019836426, "reward_meter_std": 0.0006693408940918744, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6653216481208801, "reward_total_composite_std": 0.00044622019049711525} {"timestamp_utc": "2026-04-11T22:59:44Z", "mode": "train", "global_step": 815, "epoch": 0.03273486765473752, "loss": 0.0458, "grad_norm": 8.952130317687988, "learning_rate": 7.533333333333334e-06, "num_tokens": 1806317.0, "completions/mean_length": 62.125, "completions/min_length": 54.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.125, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9308251142501831, "rewards/meter/std": 0.15261007845401764, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.8894620537757874, "rewards/total_composite/std": 0.17642506957054138, "reward": 0.8894620537757874, "reward_std": 0.1764250546693802, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024382825940847397, "sampling/sampling_logp_difference/max": 1.424285888671875, "sampling/importance_sampling_ratio/min": 0.2406802922487259, "sampling/importance_sampling_ratio/mean": 0.9998781085014343, "sampling/importance_sampling_ratio/max": 1.5671416521072388, "entropy": 0.10703426506370306, "clip_ratio/low_mean": 0.011092530796304345, "clip_ratio/low_min": 0.011092530796304345, "clip_ratio/high_mean": 0.016365812392905354, "clip_ratio/high_max": 0.016365812392905354, "clip_ratio/region_mean": 0.0274583431892097, "reward_total_mean": 0.8894620537757874, "reward_meter_mean": 0.9308251142501831, "reward_meter_std": 0.15261007845401764, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.8894620537757874, "reward_total_composite_std": 0.17642506957054138} {"timestamp_utc": "2026-04-11T22:59:49Z", "mode": "train", "global_step": 816, "epoch": 0.032775033136522476, "loss": -0.0066, "grad_norm": 8.190361022949219, "learning_rate": 7.530303030303031e-06, "num_tokens": 1807821.0, "completions/mean_length": 39.0, "completions/min_length": 38.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.0, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.7446458339691162, "rewards/meter/std": 0.30041757225990295, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7446458339691162, "rewards/total_composite/std": 0.30041757225990295, "reward": 0.7446458339691162, "reward_std": 0.30041757225990295, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.059971921145915985, "sampling/sampling_logp_difference/max": 2.114863395690918, "sampling/importance_sampling_ratio/min": 0.12064976990222931, "sampling/importance_sampling_ratio/mean": 0.9998385310173035, "sampling/importance_sampling_ratio/max": 1.8297884464263916, "entropy": 0.16819044947624207, "clip_ratio/low_mean": 0.023026316426694393, "clip_ratio/low_min": 0.023026316426694393, "clip_ratio/high_mean": 0.028449730249121785, "clip_ratio/high_max": 0.028449730249121785, "clip_ratio/region_mean": 0.05147604667581618, "reward_total_mean": 0.7446458339691162, "reward_meter_mean": 0.7446458339691162, "reward_meter_std": 0.30041757225990295, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7446458339691162, "reward_total_composite_std": 0.30041757225990295} {"timestamp_utc": "2026-04-11T22:59:54Z", "mode": "train", "global_step": 817, "epoch": 0.03281519861830743, "loss": 0.0413, "grad_norm": 3.7302210330963135, "learning_rate": 7.5272727272727274e-06, "num_tokens": 1810070.0, "completions/mean_length": 107.125, "completions/min_length": 102.0, "completions/max_length": 115.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.125, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 115.0, "rewards/meter/mean": 0.3294169008731842, "rewards/meter/std": 0.28731057047843933, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7750000357627869, "rewards/repeat_penalty/std": 0.12817399203777313, "rewards/total_composite/mean": 0.2755756676197052, "rewards/total_composite/std": 0.2410556823015213, "reward": 0.2755756676197052, "reward_std": 0.2410556823015213, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01802036352455616, "sampling/sampling_logp_difference/max": 1.4399476051330566, "sampling/importance_sampling_ratio/min": 0.2369401901960373, "sampling/importance_sampling_ratio/mean": 1.0044699907302856, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.05588802928104997, "clip_ratio/low_mean": 0.01120058260858059, "clip_ratio/low_min": 0.01120058260858059, "clip_ratio/high_mean": 0.013432800536975265, "clip_ratio/high_max": 0.013432800536975265, "clip_ratio/region_mean": 0.024633383145555854, "reward_total_mean": 0.2755756676197052, "reward_meter_mean": 0.3294169008731842, "reward_meter_std": 0.28731057047843933, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7750000357627869, "reward_repeat_penalty_std": 0.12817399203777313, "reward_total_composite_mean": 0.2755756676197052, "reward_total_composite_std": 0.2410556823015213} {"timestamp_utc": "2026-04-11T23:00:03Z", "mode": "train", "global_step": 818, "epoch": 0.032855364100092384, "loss": 0.0108, "grad_norm": 1.0928229093551636, "learning_rate": 7.524242424242425e-06, "num_tokens": 1814452.0, "completions/mean_length": 357.75, "completions/min_length": 346.0, "completions/max_length": 384.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 357.75, "completions/min_terminated_length": 346.0, "completions/max_terminated_length": 384.0, "rewards/meter/mean": 0.9965507984161377, "rewards/meter/std": 0.0037117046304047108, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.05750546231865883, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5523655414581299, "rewards/repeat_penalty/std": 0.10810358822345734, "rewards/total_composite/mean": 0.5283541083335876, "rewards/total_composite/std": 0.11482103914022446, "reward": 0.5283541083335876, "reward_std": 0.11482104659080505, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00884958729147911, "sampling/sampling_logp_difference/max": 1.062759518623352, "sampling/importance_sampling_ratio/min": 0.3455010652542114, "sampling/importance_sampling_ratio/mean": 1.0001471042633057, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03675817488692701, "clip_ratio/low_mean": 0.003796361561398953, "clip_ratio/low_min": 0.003796361561398953, "clip_ratio/high_mean": 0.006020042230375111, "clip_ratio/high_max": 0.006020042230375111, "clip_ratio/region_mean": 0.009816403791774064, "reward_total_mean": 0.5283541083335876, "reward_meter_mean": 0.9965507984161377, "reward_meter_std": 0.0037117046304047108, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.05750546231865883, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5523655414581299, "reward_repeat_penalty_std": 0.10810358822345734, "reward_total_composite_mean": 0.5283541083335876, "reward_total_composite_std": 0.11482103914022446} {"timestamp_utc": "2026-04-11T23:00:08Z", "mode": "train", "global_step": 819, "epoch": 0.03289552958187734, "loss": 0.0279, "grad_norm": 6.570438861846924, "learning_rate": 7.521212121212121e-06, "num_tokens": 1816193.0, "completions/mean_length": 61.625, "completions/min_length": 59.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.625, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9746619462966919, "rewards/meter/std": 0.020627282559871674, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9746619462966919, "rewards/total_composite/std": 0.020627282559871674, "reward": 0.9746619462966919, "reward_std": 0.020627308636903763, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02879003807902336, "sampling/sampling_logp_difference/max": 1.3429758548736572, "sampling/importance_sampling_ratio/min": 0.26106762886047363, "sampling/importance_sampling_ratio/mean": 0.9997745156288147, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10824673250317574, "clip_ratio/low_mean": 0.013770792167633772, "clip_ratio/low_min": 0.013770792167633772, "clip_ratio/high_mean": 0.022682114504277706, "clip_ratio/high_max": 0.022682114504277706, "clip_ratio/region_mean": 0.03645290667191148, "reward_total_mean": 0.9746619462966919, "reward_meter_mean": 0.9746619462966919, "reward_meter_std": 0.020627282559871674, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9746619462966919, "reward_total_composite_std": 0.020627282559871674} {"timestamp_utc": "2026-04-11T23:00:14Z", "mode": "train", "global_step": 820, "epoch": 0.03293569506366229, "loss": 0.0185, "grad_norm": 2.3457443714141846, "learning_rate": 7.518181818181819e-06, "num_tokens": 1818812.0, "completions/mean_length": 149.375, "completions/min_length": 143.0, "completions/max_length": 156.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 149.375, "completions/min_terminated_length": 143.0, "completions/max_terminated_length": 156.0, "rewards/meter/mean": 0.7805307507514954, "rewards/meter/std": 0.1253480166196823, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5714285373687744, "rewards/repeat_penalty/std": 0.17074695229530334, "rewards/total_composite/mean": 0.4300089478492737, "rewards/total_composite/std": 0.0887664407491684, "reward": 0.4300089478492737, "reward_std": 0.0887664407491684, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015637392178177834, "sampling/sampling_logp_difference/max": 1.2027087211608887, "sampling/importance_sampling_ratio/min": 0.3003794550895691, "sampling/importance_sampling_ratio/mean": 1.0010299682617188, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06724543264135718, "clip_ratio/low_mean": 0.007311595429200679, "clip_ratio/low_min": 0.007311595429200679, "clip_ratio/high_mean": 0.008506174897775054, "clip_ratio/high_max": 0.008506174897775054, "clip_ratio/region_mean": 0.015817770326975733, "reward_total_mean": 0.4300089478492737, "reward_meter_mean": 0.7805307507514954, "reward_meter_std": 0.1253480166196823, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5714285373687744, "reward_repeat_penalty_std": 0.17074695229530334, "reward_total_composite_mean": 0.4300089478492737, "reward_total_composite_std": 0.0887664407491684} {"timestamp_utc": "2026-04-11T23:00:23Z", "mode": "train", "global_step": 821, "epoch": 0.032975860545447246, "loss": -0.1471, "grad_norm": 0.9305291175842285, "learning_rate": 7.515151515151516e-06, "num_tokens": 1820662.0, "completions/mean_length": 190.25, "completions/min_length": 77.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 83.0, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.5668963193893433, "rewards/meter/std": 0.39650967717170715, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.4629100561141968, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5668963193893433, "rewards/total_composite/std": 0.39650967717170715, "reward": 0.5668963193893433, "reward_std": 0.39650964736938477, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019788384437561035, "sampling/sampling_logp_difference/max": 0.7466448545455933, "sampling/importance_sampling_ratio/min": 0.6215139627456665, "sampling/importance_sampling_ratio/mean": 1.0027402639389038, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07134009897708893, "clip_ratio/low_mean": 0.007159017724916339, "clip_ratio/low_min": 0.007159017724916339, "clip_ratio/high_mean": 0.01558885129634291, "clip_ratio/high_max": 0.01558885129634291, "clip_ratio/region_mean": 0.022747869021259248, "reward_total_mean": 0.5668963193893433, "reward_meter_mean": 0.5668963193893433, "reward_meter_std": 0.39650967717170715, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.4629100561141968, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5668963193893433, "reward_total_composite_std": 0.39650967717170715} {"timestamp_utc": "2026-04-11T23:00:28Z", "mode": "train", "global_step": 822, "epoch": 0.0330160260272322, "loss": 0.0105, "grad_norm": 3.704124689102173, "learning_rate": 7.512121212121213e-06, "num_tokens": 1822669.0, "completions/mean_length": 77.875, "completions/min_length": 75.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.875, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.8737984299659729, "rewards/meter/std": 0.14763900637626648, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8737984299659729, "rewards/total_composite/std": 0.14763900637626648, "reward": 0.8737984299659729, "reward_std": 0.14763899147510529, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018405567854642868, "sampling/sampling_logp_difference/max": 1.2704942226409912, "sampling/importance_sampling_ratio/min": 0.28069284558296204, "sampling/importance_sampling_ratio/mean": 1.0058021545410156, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09962678607553244, "clip_ratio/low_mean": 0.006382113788276911, "clip_ratio/low_min": 0.006382113788276911, "clip_ratio/high_mean": 0.009740259731188416, "clip_ratio/high_max": 0.009740259731188416, "clip_ratio/region_mean": 0.016122373519465327, "reward_total_mean": 0.8737984299659729, "reward_meter_mean": 0.8737984299659729, "reward_meter_std": 0.14763900637626648, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8737984299659729, "reward_total_composite_std": 0.14763900637626648} {"timestamp_utc": "2026-04-11T23:00:35Z", "mode": "train", "global_step": 823, "epoch": 0.033056191509017153, "loss": 0.0055, "grad_norm": 2.836785078048706, "learning_rate": 7.509090909090909e-06, "num_tokens": 1826360.0, "completions/mean_length": 284.375, "completions/min_length": 272.0, "completions/max_length": 297.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 284.375, "completions/min_terminated_length": 272.0, "completions/max_terminated_length": 297.0, "rewards/meter/mean": 0.9983997344970703, "rewards/meter/std": 0.0007759786094538867, "rewards/count_adherence/mean": 0.9821428656578064, "rewards/count_adherence/std": 0.05050762742757797, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5952796936035156, "rewards/repeat_penalty/std": 0.08133196830749512, "rewards/total_composite/mean": 0.5862284898757935, "rewards/total_composite/std": 0.09859947860240936, "reward": 0.5862284898757935, "reward_std": 0.09859946370124817, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015353661961853504, "sampling/sampling_logp_difference/max": 2.6265687942504883, "sampling/importance_sampling_ratio/min": 0.07232620567083359, "sampling/importance_sampling_ratio/mean": 1.0005059242248535, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03474967950023711, "clip_ratio/low_mean": 0.002257478976389393, "clip_ratio/low_min": 0.002257478976389393, "clip_ratio/high_mean": 0.0065078792395070195, "clip_ratio/high_max": 0.0065078792395070195, "clip_ratio/region_mean": 0.008765358215896413, "reward_total_mean": 0.5862284898757935, "reward_meter_mean": 0.9983997344970703, "reward_meter_std": 0.0007759786094538867, "reward_count_adherence_mean": 0.9821428656578064, "reward_count_adherence_std": 0.05050762742757797, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5952796936035156, "reward_repeat_penalty_std": 0.08133196830749512, "reward_total_composite_mean": 0.5862284898757935, "reward_total_composite_std": 0.09859947860240936} {"timestamp_utc": "2026-04-11T23:00:40Z", "mode": "train", "global_step": 824, "epoch": 0.03309635699080211, "loss": 0.0168, "grad_norm": 4.631432056427002, "learning_rate": 7.5060606060606065e-06, "num_tokens": 1828232.0, "completions/mean_length": 71.0, "completions/min_length": 70.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.0, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.7449524998664856, "rewards/meter/std": 0.04608132317662239, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7449524998664856, "rewards/total_composite/std": 0.04608132317662239, "reward": 0.7449524998664856, "reward_std": 0.04608132690191269, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01667422614991665, "sampling/sampling_logp_difference/max": 2.1282317638397217, "sampling/importance_sampling_ratio/min": 0.1190476045012474, "sampling/importance_sampling_ratio/mean": 0.9986546039581299, "sampling/importance_sampling_ratio/max": 1.426641583442688, "entropy": 0.04505776287987828, "clip_ratio/low_mean": 0.0017361111240461469, "clip_ratio/low_min": 0.0017361111240461469, "clip_ratio/high_mean": 0.0070436508394777775, "clip_ratio/high_max": 0.0070436508394777775, "clip_ratio/region_mean": 0.008779761963523924, "reward_total_mean": 0.7449524998664856, "reward_meter_mean": 0.7449524998664856, "reward_meter_std": 0.04608132317662239, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7449524998664856, "reward_total_composite_std": 0.04608132317662239} {"timestamp_utc": "2026-04-11T23:00:45Z", "mode": "train", "global_step": 825, "epoch": 0.03313652247258706, "loss": 0.0481, "grad_norm": 4.750690937042236, "learning_rate": 7.503030303030303e-06, "num_tokens": 1830123.0, "completions/mean_length": 68.375, "completions/min_length": 64.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.375, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.4837471544742584, "rewards/meter/std": 0.22915734350681305, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4837471544742584, "rewards/total_composite/std": 0.22915734350681305, "reward": 0.4837471544742584, "reward_std": 0.22915734350681305, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031915146857500076, "sampling/sampling_logp_difference/max": 1.8651485443115234, "sampling/importance_sampling_ratio/min": 0.15487320721149445, "sampling/importance_sampling_ratio/mean": 0.9999799728393555, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10288840066641569, "clip_ratio/low_mean": 0.024154700804501772, "clip_ratio/low_min": 0.024154700804501772, "clip_ratio/high_mean": 0.028171515092253685, "clip_ratio/high_max": 0.028171515092253685, "clip_ratio/region_mean": 0.05232621589675546, "reward_total_mean": 0.4837471544742584, "reward_meter_mean": 0.4837471544742584, "reward_meter_std": 0.22915734350681305, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.4837471544742584, "reward_total_composite_std": 0.22915734350681305} {"timestamp_utc": "2026-04-11T23:00:54Z", "mode": "train", "global_step": 826, "epoch": 0.033176687954372015, "loss": 0.0159, "grad_norm": 1.1773563623428345, "learning_rate": 7.500000000000001e-06, "num_tokens": 1835244.0, "completions/mean_length": 400.125, "completions/min_length": 375.0, "completions/max_length": 421.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 400.125, "completions/min_terminated_length": 375.0, "completions/max_terminated_length": 421.0, "rewards/meter/mean": 0.7846977710723877, "rewards/meter/std": 0.39648592472076416, "rewards/count_adherence/mean": 0.9545454978942871, "rewards/count_adherence/std": 0.0485929399728775, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5507364273071289, "rewards/repeat_penalty/std": 0.050370436161756516, "rewards/total_composite/mean": 0.3971686363220215, "rewards/total_composite/std": 0.1935662180185318, "reward": 0.3971686363220215, "reward_std": 0.1935662031173706, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007981406524777412, "sampling/sampling_logp_difference/max": 1.5345345735549927, "sampling/importance_sampling_ratio/min": 0.25283992290496826, "sampling/importance_sampling_ratio/mean": 1.0009510517120361, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0334128841641359, "clip_ratio/low_mean": 0.0033315176842734218, "clip_ratio/low_min": 0.0033315176842734218, "clip_ratio/high_mean": 0.005249909590929747, "clip_ratio/high_max": 0.005249909590929747, "clip_ratio/region_mean": 0.008581427275203168, "reward_total_mean": 0.3971686363220215, "reward_meter_mean": 0.7846977710723877, "reward_meter_std": 0.39648592472076416, "reward_count_adherence_mean": 0.9545454978942871, "reward_count_adherence_std": 0.0485929399728775, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5507364273071289, "reward_repeat_penalty_std": 0.050370436161756516, "reward_total_composite_mean": 0.3971686363220215, "reward_total_composite_std": 0.1935662180185318} {"timestamp_utc": "2026-04-11T23:00:59Z", "mode": "train", "global_step": 827, "epoch": 0.03321685343615697, "loss": -0.0012, "grad_norm": 3.7937092781066895, "learning_rate": 7.496969696969698e-06, "num_tokens": 1837278.0, "completions/mean_length": 68.25, "completions/min_length": 68.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.25, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.15423895418643951, "rewards/meter/std": 0.07883358746767044, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.15423895418643951, "rewards/total_composite/std": 0.07883358746767044, "reward": 0.15423895418643951, "reward_std": 0.07883358746767044, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016422174870967865, "sampling/sampling_logp_difference/max": 0.6457552909851074, "sampling/importance_sampling_ratio/min": 0.5242664217948914, "sampling/importance_sampling_ratio/mean": 0.9993041157722473, "sampling/importance_sampling_ratio/max": 1.73274564743042, "entropy": 0.05680654477328062, "clip_ratio/low_mean": 0.009191176504828036, "clip_ratio/low_min": 0.009191176504828036, "clip_ratio/high_mean": 0.005434782709926367, "clip_ratio/high_max": 0.005434782709926367, "clip_ratio/region_mean": 0.014625959214754403, "reward_total_mean": 0.15423895418643951, "reward_meter_mean": 0.15423895418643951, "reward_meter_std": 0.07883358746767044, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.15423895418643951, "reward_total_composite_std": 0.07883358746767044} {"timestamp_utc": "2026-04-11T23:01:04Z", "mode": "train", "global_step": 828, "epoch": 0.03325701891794192, "loss": 0.0145, "grad_norm": 6.0185723304748535, "learning_rate": 7.493939393939395e-06, "num_tokens": 1839058.0, "completions/mean_length": 63.5, "completions/min_length": 61.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.5, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9804123640060425, "rewards/meter/std": 0.014910156838595867, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9804123640060425, "rewards/total_composite/std": 0.014910156838595867, "reward": 0.9804123640060425, "reward_std": 0.014910157769918442, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028718305751681328, "sampling/sampling_logp_difference/max": 1.3754212856292725, "sampling/importance_sampling_ratio/min": 0.25273311138153076, "sampling/importance_sampling_ratio/mean": 1.0006060600280762, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11623660661280155, "clip_ratio/low_mean": 0.0038470644503831863, "clip_ratio/low_min": 0.0038470644503831863, "clip_ratio/high_mean": 0.01206992007791996, "clip_ratio/high_max": 0.01206992007791996, "clip_ratio/region_mean": 0.015916984528303146, "reward_total_mean": 0.9804123640060425, "reward_meter_mean": 0.9804123640060425, "reward_meter_std": 0.014910156838595867, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9804123640060425, "reward_total_composite_std": 0.014910156838595867} {"timestamp_utc": "2026-04-11T23:01:10Z", "mode": "train", "global_step": 829, "epoch": 0.03329718439972688, "loss": 0.0432, "grad_norm": 1.7107356786727905, "learning_rate": 7.490909090909092e-06, "num_tokens": 1841757.0, "completions/mean_length": 173.375, "completions/min_length": 151.0, "completions/max_length": 186.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 173.375, "completions/min_terminated_length": 151.0, "completions/max_terminated_length": 186.0, "rewards/meter/mean": 0.9961064457893372, "rewards/meter/std": 0.0029631657525897026, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.625, "rewards/repeat_penalty/std": 0.16967642307281494, "rewards/total_composite/mean": 0.6229597926139832, "rewards/total_composite/std": 0.17044374346733093, "reward": 0.6229597926139832, "reward_std": 0.17044374346733093, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017439143732190132, "sampling/sampling_logp_difference/max": 1.495316505432129, "sampling/importance_sampling_ratio/min": 0.22417764365673065, "sampling/importance_sampling_ratio/mean": 0.9979692101478577, "sampling/importance_sampling_ratio/max": 1.7753796577453613, "entropy": 0.05448292032815516, "clip_ratio/low_mean": 0.002016129030380398, "clip_ratio/low_min": 0.002016129030380398, "clip_ratio/high_mean": 0.016328177880495787, "clip_ratio/high_max": 0.016328177880495787, "clip_ratio/region_mean": 0.018344306910876185, "reward_total_mean": 0.6229597926139832, "reward_meter_mean": 0.9961064457893372, "reward_meter_std": 0.0029631657525897026, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.625, "reward_repeat_penalty_std": 0.16967642307281494, "reward_total_composite_mean": 0.6229597926139832, "reward_total_composite_std": 0.17044374346733093} {"timestamp_utc": "2026-04-11T23:01:15Z", "mode": "train", "global_step": 830, "epoch": 0.03333734988151183, "loss": 0.034, "grad_norm": 12.176047325134277, "learning_rate": 7.487878787878788e-06, "num_tokens": 1843710.0, "completions/mean_length": 74.125, "completions/min_length": 68.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.125, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9026094079017639, "rewards/meter/std": 0.14784321188926697, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9026094079017639, "rewards/total_composite/std": 0.14784321188926697, "reward": 0.9026094079017639, "reward_std": 0.14784321188926697, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.060764994472265244, "sampling/sampling_logp_difference/max": 5.8967156410217285, "sampling/importance_sampling_ratio/min": 0.0027484570164233446, "sampling/importance_sampling_ratio/mean": 0.9942511916160583, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12799056991934776, "clip_ratio/low_mean": 0.007117270142771304, "clip_ratio/low_min": 0.007117270142771304, "clip_ratio/high_mean": 0.019965628627687693, "clip_ratio/high_max": 0.019965628627687693, "clip_ratio/region_mean": 0.027082898770458996, "reward_total_mean": 0.9026094079017639, "reward_meter_mean": 0.9026094079017639, "reward_meter_std": 0.14784321188926697, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9026094079017639, "reward_total_composite_std": 0.14784321188926697} {"timestamp_utc": "2026-04-11T23:01:21Z", "mode": "train", "global_step": 831, "epoch": 0.033377515363296785, "loss": -0.003, "grad_norm": 4.106361389160156, "learning_rate": 7.484848484848486e-06, "num_tokens": 1845975.0, "completions/mean_length": 108.125, "completions/min_length": 105.0, "completions/max_length": 113.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.125, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 113.0, "rewards/meter/mean": 0.3786888122558594, "rewards/meter/std": 0.35290807485580444, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.3143823742866516, "rewards/total_composite/std": 0.282669335603714, "reward": 0.3143823742866516, "reward_std": 0.282669335603714, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.034197013825178146, "sampling/sampling_logp_difference/max": 2.057035207748413, "sampling/importance_sampling_ratio/min": 0.12783241271972656, "sampling/importance_sampling_ratio/mean": 0.9974236488342285, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12897953018546104, "clip_ratio/low_mean": 0.016522707068361342, "clip_ratio/low_min": 0.016522707068361342, "clip_ratio/high_mean": 0.01371849060524255, "clip_ratio/high_max": 0.01371849060524255, "clip_ratio/region_mean": 0.030241197673603892, "reward_total_mean": 0.3143823742866516, "reward_meter_mean": 0.3786888122558594, "reward_meter_std": 0.35290807485580444, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.3143823742866516, "reward_total_composite_std": 0.282669335603714} {"timestamp_utc": "2026-04-11T23:01:32Z", "mode": "train", "global_step": 832, "epoch": 0.03341768084508174, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.481818181818182e-06, "num_tokens": 1847831.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.6360248923301697, "rewards/meter/std": 0.25572946667671204, "rewards/count_adherence/mean": 0.8602941036224365, "rewards/count_adherence/std": 0.0304440688341856, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5477695465087891, "rewards/repeat_penalty/std": 0.011733362451195717, "rewards/total_composite/mean": 0.3008078336715698, "rewards/total_composite/std": 0.12469936162233353, "reward": 0.3008078336715698, "reward_std": 0.12469936162233353, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.3008078336715698, "reward_meter_mean": 0.6360248923301697, "reward_meter_std": 0.25572946667671204, "reward_count_adherence_mean": 0.8602941036224365, "reward_count_adherence_std": 0.0304440688341856, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5477695465087891, "reward_repeat_penalty_std": 0.011733362451195717, "reward_total_composite_mean": 0.3008078336715698, "reward_total_composite_std": 0.12469936162233353} {"timestamp_utc": "2026-04-11T23:01:36Z", "mode": "train", "global_step": 833, "epoch": 0.03345784632686669, "loss": 0.0048, "grad_norm": 8.433792114257812, "learning_rate": 7.47878787878788e-06, "num_tokens": 1849335.0, "completions/mean_length": 35.0, "completions/min_length": 35.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9924166798591614, "rewards/meter/std": 0.00017309709801338613, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9924166798591614, "rewards/total_composite/std": 0.00017309709801338613, "reward": 0.9924166798591614, "reward_std": 0.00017310312250629067, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008245354518294334, "sampling/sampling_logp_difference/max": 0.6460732221603394, "sampling/importance_sampling_ratio/min": 0.5240997672080994, "sampling/importance_sampling_ratio/mean": 1.0005563497543335, "sampling/importance_sampling_ratio/max": 1.1176583766937256, "entropy": 0.04187649488449097, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9924166798591614, "reward_meter_mean": 0.9924166798591614, "reward_meter_std": 0.00017309709801338613, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9924166798591614, "reward_total_composite_std": 0.00017309709801338613} {"timestamp_utc": "2026-04-11T23:01:41Z", "mode": "train", "global_step": 834, "epoch": 0.03349801180865165, "loss": 0.0342, "grad_norm": 7.6343560218811035, "learning_rate": 7.4757575757575765e-06, "num_tokens": 1851111.0, "completions/mean_length": 68.0, "completions/min_length": 62.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.40049153566360474, "rewards/meter/std": 0.3537753224372864, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.40049153566360474, "rewards/total_composite/std": 0.3537753224372864, "reward": 0.40049153566360474, "reward_std": 0.3537753224372864, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0617922842502594, "sampling/sampling_logp_difference/max": 3.229440689086914, "sampling/importance_sampling_ratio/min": 0.03957962989807129, "sampling/importance_sampling_ratio/mean": 1.0015736818313599, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.33059297781437635, "clip_ratio/low_mean": 0.035882155993022025, "clip_ratio/low_min": 0.035882155993022025, "clip_ratio/high_mean": 0.009836265817284584, "clip_ratio/high_max": 0.009836265817284584, "clip_ratio/region_mean": 0.04571842181030661, "reward_total_mean": 0.40049153566360474, "reward_meter_mean": 0.40049153566360474, "reward_meter_std": 0.3537753224372864, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.40049153566360474, "reward_total_composite_std": 0.3537753224372864} {"timestamp_utc": "2026-04-11T23:01:46Z", "mode": "train", "global_step": 835, "epoch": 0.0335381772904366, "loss": 0.0177, "grad_norm": 1.6731441020965576, "learning_rate": 7.472727272727274e-06, "num_tokens": 1853656.0, "completions/mean_length": 150.125, "completions/min_length": 149.0, "completions/max_length": 157.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 150.125, "completions/min_terminated_length": 149.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.9238545894622803, "rewards/meter/std": 0.2039637267589569, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6159030199050903, "rewards/total_composite/std": 0.13597580790519714, "reward": 0.6159030199050903, "reward_std": 0.13597580790519714, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00628118310123682, "sampling/sampling_logp_difference/max": 1.1030478477478027, "sampling/importance_sampling_ratio/min": 0.3318580687046051, "sampling/importance_sampling_ratio/mean": 0.9989848136901855, "sampling/importance_sampling_ratio/max": 1.6532901525497437, "entropy": 0.03071731375530362, "clip_ratio/low_mean": 0.0007961783558130264, "clip_ratio/low_min": 0.0007961783558130264, "clip_ratio/high_mean": 0.004172259592451155, "clip_ratio/high_max": 0.004172259592451155, "clip_ratio/region_mean": 0.004968437948264182, "reward_total_mean": 0.6159030199050903, "reward_meter_mean": 0.9238545894622803, "reward_meter_std": 0.2039637267589569, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6159030199050903, "reward_total_composite_std": 0.13597580790519714} {"timestamp_utc": "2026-04-11T23:01:52Z", "mode": "train", "global_step": 836, "epoch": 0.033578342772221555, "loss": -0.0007, "grad_norm": 2.8023297786712646, "learning_rate": 7.46969696969697e-06, "num_tokens": 1856194.0, "completions/mean_length": 142.25, "completions/min_length": 139.0, "completions/max_length": 150.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 142.25, "completions/min_terminated_length": 139.0, "completions/max_terminated_length": 150.0, "rewards/meter/mean": 0.8496776819229126, "rewards/meter/std": 0.3326415419578552, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8214285373687744, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.7131145596504211, "rewards/total_composite/std": 0.2920505106449127, "reward": 0.7131145596504211, "reward_std": 0.29205048084259033, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012105772271752357, "sampling/sampling_logp_difference/max": 1.3084237575531006, "sampling/importance_sampling_ratio/min": 0.2702457010746002, "sampling/importance_sampling_ratio/mean": 0.9984580278396606, "sampling/importance_sampling_ratio/max": 1.5496021509170532, "entropy": 0.044805840123444796, "clip_ratio/low_mean": 0.001773406460415572, "clip_ratio/low_min": 0.001773406460415572, "clip_ratio/high_mean": 0.007069471699651331, "clip_ratio/high_max": 0.007069471699651331, "clip_ratio/region_mean": 0.008842878160066903, "reward_total_mean": 0.7131145596504211, "reward_meter_mean": 0.8496776819229126, "reward_meter_std": 0.3326415419578552, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8214285373687744, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.7131145596504211, "reward_total_composite_std": 0.2920505106449127} {"timestamp_utc": "2026-04-11T23:01:56Z", "mode": "train", "global_step": 837, "epoch": 0.03361850825400651, "loss": 0.0053, "grad_norm": 11.809463500976562, "learning_rate": 7.4666666666666675e-06, "num_tokens": 1857618.0, "completions/mean_length": 33.0, "completions/min_length": 33.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9952504634857178, "rewards/meter/std": 0.0004852505517192185, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9952504634857178, "rewards/total_composite/std": 0.0004852505517192185, "reward": 0.9952504634857178, "reward_std": 0.0004852571291849017, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017783869057893753, "sampling/sampling_logp_difference/max": 0.7500922679901123, "sampling/importance_sampling_ratio/min": 0.4723230004310608, "sampling/importance_sampling_ratio/mean": 1.0051133632659912, "sampling/importance_sampling_ratio/max": 1.5762113332748413, "entropy": 0.0944258663803339, "clip_ratio/low_mean": 0.01515151560306549, "clip_ratio/low_min": 0.01515151560306549, "clip_ratio/high_mean": 0.011363636702299118, "clip_ratio/high_max": 0.011363636702299118, "clip_ratio/region_mean": 0.02651515230536461, "reward_total_mean": 0.9952504634857178, "reward_meter_mean": 0.9952504634857178, "reward_meter_std": 0.0004852505517192185, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9952504634857178, "reward_total_composite_std": 0.0004852505517192185} {"timestamp_utc": "2026-04-11T23:02:04Z", "mode": "train", "global_step": 838, "epoch": 0.03365867373579146, "loss": 0.0061, "grad_norm": 2.168991804122925, "learning_rate": 7.463636363636364e-06, "num_tokens": 1861717.0, "completions/mean_length": 295.375, "completions/min_length": 264.0, "completions/max_length": 331.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 295.375, "completions/min_terminated_length": 264.0, "completions/max_terminated_length": 331.0, "rewards/meter/mean": 0.869155764579773, "rewards/meter/std": 0.35062167048454285, "rewards/count_adherence/mean": 0.828125, "rewards/count_adherence/std": 0.06469365209341049, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.5626935362815857, "rewards/repeat_penalty/std": 0.04896574467420578, "rewards/total_composite/mean": 0.4079222083091736, "rewards/total_composite/std": 0.1708296537399292, "reward": 0.4079222083091736, "reward_std": 0.170829638838768, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009027719497680664, "sampling/sampling_logp_difference/max": 1.8500938415527344, "sampling/importance_sampling_ratio/min": 0.15722240507602692, "sampling/importance_sampling_ratio/mean": 0.9985241889953613, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.02827131818048656, "clip_ratio/low_mean": 0.0017137863032985479, "clip_ratio/low_min": 0.0017137863032985479, "clip_ratio/high_mean": 0.004315092111937702, "clip_ratio/high_max": 0.004315092111937702, "clip_ratio/region_mean": 0.0060288784152362496, "reward_total_mean": 0.4079222083091736, "reward_meter_mean": 0.869155764579773, "reward_meter_std": 0.35062167048454285, "reward_count_adherence_mean": 0.828125, "reward_count_adherence_std": 0.06469365209341049, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.5626935362815857, "reward_repeat_penalty_std": 0.04896574467420578, "reward_total_composite_mean": 0.4079222083091736, "reward_total_composite_std": 0.1708296537399292} {"timestamp_utc": "2026-04-11T23:02:09Z", "mode": "train", "global_step": 839, "epoch": 0.033698839217576416, "loss": 0.0131, "grad_norm": 2.164283275604248, "learning_rate": 7.460606060606061e-06, "num_tokens": 1863934.0, "completions/mean_length": 112.125, "completions/min_length": 109.0, "completions/max_length": 119.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.125, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 119.0, "rewards/meter/mean": 0.9943764209747314, "rewards/meter/std": 0.0010588886216282845, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.795501172542572, "rewards/total_composite/std": 0.0008471080800518394, "reward": 0.795501172542572, "reward_std": 0.0008471118635497987, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008892908692359924, "sampling/sampling_logp_difference/max": 0.7849429249763489, "sampling/importance_sampling_ratio/min": 0.4561457335948944, "sampling/importance_sampling_ratio/mean": 1.0025136470794678, "sampling/importance_sampling_ratio/max": 1.7390761375427246, "entropy": 0.047560357954353094, "clip_ratio/low_mean": 0.006605345057323575, "clip_ratio/low_min": 0.006605345057323575, "clip_ratio/high_mean": 0.0022935778833925724, "clip_ratio/high_max": 0.0022935778833925724, "clip_ratio/region_mean": 0.008898922940716147, "reward_total_mean": 0.795501172542572, "reward_meter_mean": 0.9943764209747314, "reward_meter_std": 0.0010588886216282845, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.795501172542572, "reward_total_composite_std": 0.0008471080800518394} {"timestamp_utc": "2026-04-11T23:02:14Z", "mode": "train", "global_step": 840, "epoch": 0.03373900469936137, "loss": 0.036, "grad_norm": 5.7492289543151855, "learning_rate": 7.4575757575757575e-06, "num_tokens": 1865375.0, "completions/mean_length": 37.125, "completions/min_length": 36.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.125, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.7395036816596985, "rewards/meter/std": 0.2953052818775177, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7395036816596985, "rewards/total_composite/std": 0.2953052818775177, "reward": 0.7395036816596985, "reward_std": 0.2953052818775177, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01924203522503376, "sampling/sampling_logp_difference/max": 0.40451645851135254, "sampling/importance_sampling_ratio/min": 0.6779606938362122, "sampling/importance_sampling_ratio/mean": 1.0085185766220093, "sampling/importance_sampling_ratio/max": 1.4985777139663696, "entropy": 0.11148730758577585, "clip_ratio/low_mean": 0.006410256493836641, "clip_ratio/low_min": 0.006410256493836641, "clip_ratio/high_mean": 0.010228979168459773, "clip_ratio/high_max": 0.010228979168459773, "clip_ratio/region_mean": 0.016639235662296414, "reward_total_mean": 0.7395036816596985, "reward_meter_mean": 0.7395036816596985, "reward_meter_std": 0.2953052818775177, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7395036816596985, "reward_total_composite_std": 0.2953052818775177} {"timestamp_utc": "2026-04-11T23:02:19Z", "mode": "train", "global_step": 841, "epoch": 0.033779170181146324, "loss": -0.0141, "grad_norm": 3.028710126876831, "learning_rate": 7.454545454545456e-06, "num_tokens": 1867245.0, "completions/mean_length": 64.75, "completions/min_length": 64.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.75, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9528952240943909, "rewards/meter/std": 0.01815449446439743, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9528952240943909, "rewards/total_composite/std": 0.01815449446439743, "reward": 0.9528952240943909, "reward_std": 0.018154479563236237, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01176523882895708, "sampling/sampling_logp_difference/max": 0.8644721508026123, "sampling/importance_sampling_ratio/min": 0.5180370807647705, "sampling/importance_sampling_ratio/mean": 1.0040351152420044, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.045040544122457504, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/high_mean": 0.0018656715983524919, "clip_ratio/high_max": 0.0018656715983524919, "clip_ratio/region_mean": 0.005771921598352492, "reward_total_mean": 0.9528952240943909, "reward_meter_mean": 0.9528952240943909, "reward_meter_std": 0.01815449446439743, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9528952240943909, "reward_total_composite_std": 0.01815449446439743} {"timestamp_utc": "2026-04-11T23:02:24Z", "mode": "train", "global_step": 842, "epoch": 0.03381933566293128, "loss": 0.0297, "grad_norm": 9.17272663116455, "learning_rate": 7.451515151515152e-06, "num_tokens": 1868749.0, "completions/mean_length": 37.0, "completions/min_length": 36.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.0, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.700322687625885, "rewards/meter/std": 0.3828321695327759, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.700322687625885, "rewards/total_composite/std": 0.3828321695327759, "reward": 0.700322687625885, "reward_std": 0.3828321397304535, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011855069547891617, "sampling/sampling_logp_difference/max": 0.964850664138794, "sampling/importance_sampling_ratio/min": 0.6239942312240601, "sampling/importance_sampling_ratio/mean": 1.0074673891067505, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.04681507125496864, "clip_ratio/low_mean": 0.009699730202555656, "clip_ratio/low_min": 0.009699730202555656, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.009699730202555656, "reward_total_mean": 0.700322687625885, "reward_meter_mean": 0.700322687625885, "reward_meter_std": 0.3828321695327759, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.700322687625885, "reward_total_composite_std": 0.3828321695327759} {"timestamp_utc": "2026-04-11T23:02:33Z", "mode": "train", "global_step": 843, "epoch": 0.03385950114471623, "loss": 0.0211, "grad_norm": 2.23907470703125, "learning_rate": 7.448484848484849e-06, "num_tokens": 1874562.0, "completions/mean_length": 452.625, "completions/min_length": 426.0, "completions/max_length": 461.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 452.625, "completions/min_terminated_length": 426.0, "completions/max_terminated_length": 461.0, "rewards/meter/mean": 0.9962563514709473, "rewards/meter/std": 0.001195572316646576, "rewards/count_adherence/mean": 0.9270833730697632, "rewards/count_adherence/std": 0.029462777078151703, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5606521368026733, "rewards/repeat_penalty/std": 0.0018446200992912054, "rewards/total_composite/mean": 0.5178769826889038, "rewards/total_composite/std": 0.01842796988785267, "reward": 0.5178769826889038, "reward_std": 0.018427973613142967, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004932690877467394, "sampling/sampling_logp_difference/max": 4.1077117919921875, "sampling/importance_sampling_ratio/min": 0.01644536294043064, "sampling/importance_sampling_ratio/mean": 0.999667763710022, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.01467959355795756, "clip_ratio/low_mean": 0.0011018531513400376, "clip_ratio/low_min": 0.0011018531513400376, "clip_ratio/high_mean": 0.0002934272342827171, "clip_ratio/high_max": 0.0002934272342827171, "clip_ratio/region_mean": 0.0013952803856227547, "reward_total_mean": 0.5178769826889038, "reward_meter_mean": 0.9962563514709473, "reward_meter_std": 0.001195572316646576, "reward_count_adherence_mean": 0.9270833730697632, "reward_count_adherence_std": 0.029462777078151703, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5606521368026733, "reward_repeat_penalty_std": 0.0018446200992912054, "reward_total_composite_mean": 0.5178769826889038, "reward_total_composite_std": 0.01842796988785267} {"timestamp_utc": "2026-04-11T23:02:39Z", "mode": "train", "global_step": 844, "epoch": 0.033899666626501186, "loss": -0.0116, "grad_norm": 2.640986919403076, "learning_rate": 7.445454545454546e-06, "num_tokens": 1877374.0, "completions/mean_length": 149.5, "completions/min_length": 144.0, "completions/max_length": 158.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 149.5, "completions/min_terminated_length": 144.0, "completions/max_terminated_length": 158.0, "rewards/meter/mean": 0.5467606782913208, "rewards/meter/std": 0.4782540798187256, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.3916763961315155, "rewards/total_composite/std": 0.34031400084495544, "reward": 0.3916763961315155, "reward_std": 0.34031397104263306, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019425617530941963, "sampling/sampling_logp_difference/max": 3.396358013153076, "sampling/importance_sampling_ratio/min": 0.03349503502249718, "sampling/importance_sampling_ratio/mean": 1.0008183717727661, "sampling/importance_sampling_ratio/max": 1.7641745805740356, "entropy": 0.09837253391742706, "clip_ratio/low_mean": 0.011086393264122307, "clip_ratio/low_min": 0.011086393264122307, "clip_ratio/high_mean": 0.002457805967424065, "clip_ratio/high_max": 0.002457805967424065, "clip_ratio/region_mean": 0.013544199231546372, "reward_total_mean": 0.3916763961315155, "reward_meter_mean": 0.5467606782913208, "reward_meter_std": 0.4782540798187256, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.3916763961315155, "reward_total_composite_std": 0.34031400084495544} {"timestamp_utc": "2026-04-11T23:02:44Z", "mode": "train", "global_step": 845, "epoch": 0.03393983210828614, "loss": 0.0112, "grad_norm": 6.265551567077637, "learning_rate": 7.442424242424243e-06, "num_tokens": 1879205.0, "completions/mean_length": 71.875, "completions/min_length": 68.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.875, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.522850513458252, "rewards/meter/std": 0.4167693555355072, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.5038976669311523, "rewards/total_composite/std": 0.42369261384010315, "reward": 0.5038976669311523, "reward_std": 0.42369258403778076, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03976872190833092, "sampling/sampling_logp_difference/max": 1.2724390029907227, "sampling/importance_sampling_ratio/min": 0.2801474928855896, "sampling/importance_sampling_ratio/mean": 1.0023622512817383, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.16639000456780195, "clip_ratio/low_mean": 0.01767799479421228, "clip_ratio/low_min": 0.01767799479421228, "clip_ratio/high_mean": 0.017289764247834682, "clip_ratio/high_max": 0.017289764247834682, "clip_ratio/region_mean": 0.034967759042046964, "reward_total_mean": 0.5038976669311523, "reward_meter_mean": 0.522850513458252, "reward_meter_std": 0.4167693555355072, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.5038976669311523, "reward_total_composite_std": 0.42369261384010315} {"timestamp_utc": "2026-04-11T23:02:52Z", "mode": "train", "global_step": 846, "epoch": 0.033979997590071094, "loss": -0.0053, "grad_norm": 4.094456672668457, "learning_rate": 7.439393939393939e-06, "num_tokens": 1883732.0, "completions/mean_length": 321.875, "completions/min_length": 309.0, "completions/max_length": 336.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 321.875, "completions/min_terminated_length": 309.0, "completions/max_terminated_length": 336.0, "rewards/meter/mean": 0.6540272831916809, "rewards/meter/std": 0.39356982707977295, "rewards/count_adherence/mean": 0.9861111044883728, "rewards/count_adherence/std": 0.03928370773792267, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4677932560443878, "rewards/repeat_penalty/std": 0.22017233073711395, "rewards/total_composite/mean": 0.3497573733329773, "rewards/total_composite/std": 0.2734247148036957, "reward": 0.3497573733329773, "reward_std": 0.2734247148036957, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007702399045228958, "sampling/sampling_logp_difference/max": 1.841371774673462, "sampling/importance_sampling_ratio/min": 0.15859971940517426, "sampling/importance_sampling_ratio/mean": 0.9985901713371277, "sampling/importance_sampling_ratio/max": 1.768311619758606, "entropy": 0.03188518015667796, "clip_ratio/low_mean": 0.0011160714784637094, "clip_ratio/low_min": 0.0011160714784637094, "clip_ratio/high_mean": 0.0058011687360703945, "clip_ratio/high_max": 0.0058011687360703945, "clip_ratio/region_mean": 0.006917240214534104, "reward_total_mean": 0.3497573733329773, "reward_meter_mean": 0.6540272831916809, "reward_meter_std": 0.39356982707977295, "reward_count_adherence_mean": 0.9861111044883728, "reward_count_adherence_std": 0.03928370773792267, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4677932560443878, "reward_repeat_penalty_std": 0.22017233073711395, "reward_total_composite_mean": 0.3497573733329773, "reward_total_composite_std": 0.2734247148036957} {"timestamp_utc": "2026-04-11T23:02:57Z", "mode": "train", "global_step": 847, "epoch": 0.03402016307185605, "loss": 0.045, "grad_norm": 6.007223606109619, "learning_rate": 7.4363636363636375e-06, "num_tokens": 1885713.0, "completions/mean_length": 72.625, "completions/min_length": 68.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.625, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.26388847827911377, "rewards/meter/std": 0.3744986653327942, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.2634860873222351, "rewards/total_composite/std": 0.3748124837875366, "reward": 0.2634860873222351, "reward_std": 0.3748124837875366, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04990806803107262, "sampling/sampling_logp_difference/max": 1.3789244890213013, "sampling/importance_sampling_ratio/min": 0.2518492639064789, "sampling/importance_sampling_ratio/mean": 1.007526159286499, "sampling/importance_sampling_ratio/max": 1.518433690071106, "entropy": 0.2849195022135973, "clip_ratio/low_mean": 0.02969917980954051, "clip_ratio/low_min": 0.02969917980954051, "clip_ratio/high_mean": 0.019930581096559763, "clip_ratio/high_max": 0.019930581096559763, "clip_ratio/region_mean": 0.04962976090610027, "reward_total_mean": 0.2634860873222351, "reward_meter_mean": 0.26388847827911377, "reward_meter_std": 0.3744986653327942, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.2634860873222351, "reward_total_composite_std": 0.3748124837875366} {"timestamp_utc": "2026-04-11T23:03:02Z", "mode": "train", "global_step": 848, "epoch": 0.034060328553641, "loss": 0.0001, "grad_norm": 2.973163366317749, "learning_rate": 7.433333333333334e-06, "num_tokens": 1888010.0, "completions/mean_length": 106.125, "completions/min_length": 100.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.125, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.6358721256256104, "rewards/meter/std": 0.32246294617652893, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7749999761581421, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.5016140341758728, "rewards/total_composite/std": 0.2674255073070526, "reward": 0.5016140341758728, "reward_std": 0.2674255073070526, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022296659648418427, "sampling/sampling_logp_difference/max": 1.8602898120880127, "sampling/importance_sampling_ratio/min": 0.1556275188922882, "sampling/importance_sampling_ratio/mean": 0.9971781969070435, "sampling/importance_sampling_ratio/max": 1.6591869592666626, "entropy": 0.08717937674373388, "clip_ratio/low_mean": 0.005942982388660312, "clip_ratio/low_min": 0.005942982388660312, "clip_ratio/high_mean": 0.013045326224528253, "clip_ratio/high_max": 0.013045326224528253, "clip_ratio/region_mean": 0.018988308613188565, "reward_total_mean": 0.5016140341758728, "reward_meter_mean": 0.6358721256256104, "reward_meter_std": 0.32246294617652893, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7749999761581421, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.5016140341758728, "reward_total_composite_std": 0.2674255073070526} {"timestamp_utc": "2026-04-11T23:03:10Z", "mode": "train", "global_step": 849, "epoch": 0.034100494035425956, "loss": 0.001, "grad_norm": 0.23050236701965332, "learning_rate": 7.430303030303031e-06, "num_tokens": 1892024.0, "completions/mean_length": 283.75, "completions/min_length": 283.0, "completions/max_length": 284.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 283.75, "completions/min_terminated_length": 283.0, "completions/max_terminated_length": 284.0, "rewards/meter/mean": 0.9965326189994812, "rewards/meter/std": 0.00021112659305799752, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6000000238418579, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5979195833206177, "rewards/total_composite/std": 0.00012666928523685783, "reward": 0.5979195833206177, "reward_std": 0.000126663115224801, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0016167466528713703, "sampling/sampling_logp_difference/max": 0.7609295845031738, "sampling/importance_sampling_ratio/min": 0.4672318696975708, "sampling/importance_sampling_ratio/mean": 1.0001165866851807, "sampling/importance_sampling_ratio/max": 1.2486354112625122, "entropy": 0.010985135799273849, "clip_ratio/low_mean": 0.00044014083687216043, "clip_ratio/low_min": 0.00044014083687216043, "clip_ratio/high_mean": 0.00044169611646793783, "clip_ratio/high_max": 0.00044169611646793783, "clip_ratio/region_mean": 0.0008818369533400983, "reward_total_mean": 0.5979195833206177, "reward_meter_mean": 0.9965326189994812, "reward_meter_std": 0.00021112659305799752, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6000000238418579, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5979195833206177, "reward_total_composite_std": 0.00012666928523685783} {"timestamp_utc": "2026-04-11T23:03:15Z", "mode": "train", "global_step": 850, "epoch": 0.03414065951721091, "loss": -0.0004, "grad_norm": 7.60121488571167, "learning_rate": 7.4272727272727275e-06, "num_tokens": 1893564.0, "completions/mean_length": 35.5, "completions/min_length": 33.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.5, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9917958974838257, "rewards/meter/std": 0.008829712867736816, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8676612377166748, "rewards/total_composite/std": 0.35069888830184937, "reward": 0.8676612377166748, "reward_std": 0.350698858499527, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04963725060224533, "sampling/sampling_logp_difference/max": 3.3897039890289307, "sampling/importance_sampling_ratio/min": 0.03371865674853325, "sampling/importance_sampling_ratio/mean": 1.0052940845489502, "sampling/importance_sampling_ratio/max": 1.9962736368179321, "entropy": 0.15574337635189295, "clip_ratio/low_mean": 0.0071428571827709675, "clip_ratio/low_min": 0.0071428571827709675, "clip_ratio/high_mean": 0.031166881788522005, "clip_ratio/high_max": 0.031166881788522005, "clip_ratio/region_mean": 0.03830973897129297, "reward_total_mean": 0.8676612377166748, "reward_meter_mean": 0.9917958974838257, "reward_meter_std": 0.008829712867736816, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8676612377166748, "reward_total_composite_std": 0.35069888830184937} {"timestamp_utc": "2026-04-11T23:04:30Z", "mode": "eval", "global_step": 850, "epoch": 0.03414065951721091, "eval_loss": NaN, "eval_runtime": 75.7065, "eval_samples_per_second": 1.374, "eval_steps_per_second": 0.172, "eval_num_tokens": 1893564.0, "eval_completions/mean_length": 209.85576923076923, "eval_completions/min_length": 62.0, "eval_completions/max_length": 402.6923076923077, "eval_completions/clipped_ratio": 0.009615384615384616, "eval_completions/mean_terminated_length": 206.50412104679987, "eval_completions/min_terminated_length": 62.0, "eval_completions/max_terminated_length": 389.9230769230769, "eval_rewards/meter/mean": 0.6430152883896461, "eval_rewards/meter/std": 0.3854568119232471, "eval_rewards/count_adherence/mean": 0.9535174920008733, "eval_rewards/count_adherence/std": 0.07147554422800358, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/repeat_penalty/mean": 0.6578034116671636, "eval_rewards/repeat_penalty/std": 0.21721924497531012, "eval_rewards/total_composite/mean": 0.4121822485556969, "eval_rewards/total_composite/std": 0.3158151931487597, "eval_reward": 0.4121822485556969, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.00518767936871602, "eval_sampling/sampling_logp_difference/max": 0.7618373265633216, "eval_sampling/importance_sampling_ratio/min": 0.4886489510536194, "eval_sampling/importance_sampling_ratio/mean": 1.001105854144463, "eval_sampling/importance_sampling_ratio/max": 1.308758864035973, "eval_entropy": 0.046050483074325785, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.4121822485556969, "eval_reward_meter_mean": 0.6430152883896461, "eval_reward_meter_std": 0.3854568119232471, "eval_reward_count_adherence_mean": 0.9535174920008733, "eval_reward_count_adherence_std": 0.07147554422800358, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_repeat_penalty_mean": 0.6578034116671636, "eval_reward_repeat_penalty_std": 0.21721924497531012, "eval_reward_total_composite_mean": 0.4121822485556969, "eval_reward_total_composite_std": 0.3158151931487597} {"timestamp_utc": "2026-04-11T23:04:39Z", "mode": "train", "global_step": 851, "epoch": 0.034180824998995864, "loss": 0.0004, "grad_norm": 2.013568162918091, "learning_rate": 7.424242424242425e-06, "num_tokens": 1896556.0, "completions/mean_length": 186.0, "completions/min_length": 178.0, "completions/max_length": 199.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 186.0, "completions/min_terminated_length": 178.0, "completions/max_terminated_length": 199.0, "rewards/meter/mean": 0.989942193031311, "rewards/meter/std": 0.007896743714809418, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5795454382896423, "rewards/repeat_penalty/std": 0.16070608794689178, "rewards/total_composite/mean": 0.5742474794387817, "rewards/total_composite/std": 0.16005170345306396, "reward": 0.5742474794387817, "reward_std": 0.16005171835422516, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011838030070066452, "sampling/sampling_logp_difference/max": 1.5690075159072876, "sampling/importance_sampling_ratio/min": 0.3101670742034912, "sampling/importance_sampling_ratio/mean": 1.0001335144042969, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.033908116864040494, "clip_ratio/low_mean": 0.0013513513840734959, "clip_ratio/low_min": 0.0013513513840734959, "clip_ratio/high_mean": 0.00752471067244187, "clip_ratio/high_max": 0.00752471067244187, "clip_ratio/region_mean": 0.008876062056515366, "reward_total_mean": 0.5742474794387817, "reward_meter_mean": 0.989942193031311, "reward_meter_std": 0.007896743714809418, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5795454382896423, "reward_repeat_penalty_std": 0.16070608794689178, "reward_total_composite_mean": 0.5742474794387817, "reward_total_composite_std": 0.16005170345306396} {"timestamp_utc": "2026-04-11T23:04:44Z", "mode": "train", "global_step": 852, "epoch": 0.03422099048078082, "loss": -0.0116, "grad_norm": 6.178599834442139, "learning_rate": 7.421212121212121e-06, "num_tokens": 1898384.0, "completions/mean_length": 72.5, "completions/min_length": 68.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.5, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.6383301019668579, "rewards/meter/std": 0.34624141454696655, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.6031950116157532, "rewards/total_composite/std": 0.3757425546646118, "reward": 0.6031950116157532, "reward_std": 0.3757425546646118, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03135836869478226, "sampling/sampling_logp_difference/max": 2.203850746154785, "sampling/importance_sampling_ratio/min": 0.11037729680538177, "sampling/importance_sampling_ratio/mean": 0.9961123466491699, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10962006729096174, "clip_ratio/low_mean": 0.006956097553484142, "clip_ratio/low_min": 0.006956097553484142, "clip_ratio/high_mean": 0.006779896444641054, "clip_ratio/high_max": 0.006779896444641054, "clip_ratio/region_mean": 0.013735993998125196, "reward_total_mean": 0.6031950116157532, "reward_meter_mean": 0.6383301019668579, "reward_meter_std": 0.34624141454696655, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_total_composite_mean": 0.6031950116157532, "reward_total_composite_std": 0.3757425546646118} {"timestamp_utc": "2026-04-11T23:04:53Z", "mode": "train", "global_step": 853, "epoch": 0.03426115596256577, "loss": 0.0349, "grad_norm": 1.2332594394683838, "learning_rate": 7.4181818181818185e-06, "num_tokens": 1903150.0, "completions/mean_length": 373.75, "completions/min_length": 351.0, "completions/max_length": 405.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 373.75, "completions/min_terminated_length": 351.0, "completions/max_terminated_length": 405.0, "rewards/meter/mean": 0.6268174648284912, "rewards/meter/std": 0.39189305901527405, "rewards/count_adherence/mean": 0.7692307829856873, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5203947424888611, "rewards/repeat_penalty/std": 0.10010629892349243, "rewards/total_composite/mean": 0.2684391736984253, "rewards/total_composite/std": 0.18215535581111908, "reward": 0.2684391736984253, "reward_std": 0.18215534090995789, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00892745703458786, "sampling/sampling_logp_difference/max": 1.788743495941162, "sampling/importance_sampling_ratio/min": 0.1743081957101822, "sampling/importance_sampling_ratio/mean": 1.0005637407302856, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03790131723508239, "clip_ratio/low_mean": 0.002888624498154968, "clip_ratio/low_min": 0.002888624498154968, "clip_ratio/high_mean": 0.0017242564936168492, "clip_ratio/high_max": 0.0017242564936168492, "clip_ratio/region_mean": 0.004612880991771817, "reward_total_mean": 0.2684391736984253, "reward_meter_mean": 0.6268174648284912, "reward_meter_std": 0.39189305901527405, "reward_count_adherence_mean": 0.7692307829856873, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5203947424888611, "reward_repeat_penalty_std": 0.10010629892349243, "reward_total_composite_mean": 0.2684391736984253, "reward_total_composite_std": 0.18215535581111908} {"timestamp_utc": "2026-04-11T23:04:58Z", "mode": "train", "global_step": 854, "epoch": 0.034301321444350726, "loss": 0.023, "grad_norm": 3.387570858001709, "learning_rate": 7.415151515151515e-06, "num_tokens": 1905516.0, "completions/mean_length": 125.75, "completions/min_length": 118.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 125.75, "completions/min_terminated_length": 118.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.9104515314102173, "rewards/meter/std": 0.12032636255025864, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6607142686843872, "rewards/repeat_penalty/std": 0.15152288973331451, "rewards/total_composite/mean": 0.5973337292671204, "rewards/total_composite/std": 0.1518002152442932, "reward": 0.5973337292671204, "reward_std": 0.1518002152442932, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014981008134782314, "sampling/sampling_logp_difference/max": 2.355228900909424, "sampling/importance_sampling_ratio/min": 0.09487178921699524, "sampling/importance_sampling_ratio/mean": 0.9986640214920044, "sampling/importance_sampling_ratio/max": 1.784404993057251, "entropy": 0.06485871458426118, "clip_ratio/low_mean": 0.0039122195448726416, "clip_ratio/low_min": 0.0039122195448726416, "clip_ratio/high_mean": 0.008279926958493888, "clip_ratio/high_max": 0.008279926958493888, "clip_ratio/region_mean": 0.01219214650336653, "reward_total_mean": 0.5973337292671204, "reward_meter_mean": 0.9104515314102173, "reward_meter_std": 0.12032636255025864, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6607142686843872, "reward_repeat_penalty_std": 0.15152288973331451, "reward_total_composite_mean": 0.5973337292671204, "reward_total_composite_std": 0.1518002152442932} {"timestamp_utc": "2026-04-11T23:05:03Z", "mode": "train", "global_step": 855, "epoch": 0.03434148692613568, "loss": -0.006, "grad_norm": 5.122466564178467, "learning_rate": 7.412121212121213e-06, "num_tokens": 1907251.0, "completions/mean_length": 64.875, "completions/min_length": 63.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.875, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9911971092224121, "rewards/meter/std": 0.0026730033569037914, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9497148990631104, "rewards/total_composite/std": 0.11557892709970474, "reward": 0.9497148990631104, "reward_std": 0.11557891964912415, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021537287160754204, "sampling/sampling_logp_difference/max": 1.6212968826293945, "sampling/importance_sampling_ratio/min": 0.32319414615631104, "sampling/importance_sampling_ratio/mean": 0.9993307590484619, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06852800818160176, "clip_ratio/low_mean": 0.003968254197388887, "clip_ratio/low_min": 0.003968254197388887, "clip_ratio/high_mean": 0.01156850962433964, "clip_ratio/high_max": 0.01156850962433964, "clip_ratio/region_mean": 0.015536763821728528, "reward_total_mean": 0.9497148990631104, "reward_meter_mean": 0.9911971092224121, "reward_meter_std": 0.0026730033569037914, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.9497148990631104, "reward_total_composite_std": 0.11557892709970474} {"timestamp_utc": "2026-04-11T23:05:08Z", "mode": "train", "global_step": 856, "epoch": 0.03438165240792063, "loss": 0.0002, "grad_norm": 6.793356418609619, "learning_rate": 7.40909090909091e-06, "num_tokens": 1909129.0, "completions/mean_length": 64.75, "completions/min_length": 63.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.75, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9881373643875122, "rewards/meter/std": 0.009240454062819481, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9881373643875122, "rewards/total_composite/std": 0.009240454062819481, "reward": 0.9881373643875122, "reward_std": 0.00924046989530325, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017327219247817993, "sampling/sampling_logp_difference/max": 1.138974666595459, "sampling/importance_sampling_ratio/min": 0.3201471269130707, "sampling/importance_sampling_ratio/mean": 0.9964679479598999, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07705914694815874, "clip_ratio/low_mean": 0.001923076924867928, "clip_ratio/low_min": 0.001923076924867928, "clip_ratio/high_mean": 0.011599511839449406, "clip_ratio/high_max": 0.011599511839449406, "clip_ratio/region_mean": 0.013522588764317334, "reward_total_mean": 0.9881373643875122, "reward_meter_mean": 0.9881373643875122, "reward_meter_std": 0.009240454062819481, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9881373643875122, "reward_total_composite_std": 0.009240454062819481} {"timestamp_utc": "2026-04-11T23:05:13Z", "mode": "train", "global_step": 857, "epoch": 0.03442181788970559, "loss": 0.0072, "grad_norm": 3.6093900203704834, "learning_rate": 7.406060606060607e-06, "num_tokens": 1910935.0, "completions/mean_length": 75.75, "completions/min_length": 68.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.75, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.991020679473877, "rewards/meter/std": 0.00623312359675765, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9496715664863586, "rewards/total_composite/std": 0.11656945198774338, "reward": 0.9496715664863586, "reward_std": 0.11656944453716278, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.041222382336854935, "sampling/sampling_logp_difference/max": 3.455476760864258, "sampling/importance_sampling_ratio/min": 0.03157224878668785, "sampling/importance_sampling_ratio/mean": 0.9909502863883972, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07773020630702376, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.018700583139434457, "clip_ratio/high_max": 0.018700583139434457, "clip_ratio/region_mean": 0.018700583139434457, "reward_total_mean": 0.9496715664863586, "reward_meter_mean": 0.991020679473877, "reward_meter_std": 0.00623312359675765, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.9496715664863586, "reward_total_composite_std": 0.11656945198774338} {"timestamp_utc": "2026-04-11T23:05:20Z", "mode": "train", "global_step": 858, "epoch": 0.03446198337149054, "loss": 0.0285, "grad_norm": 1.946842908859253, "learning_rate": 7.403030303030304e-06, "num_tokens": 1914420.0, "completions/mean_length": 254.625, "completions/min_length": 233.0, "completions/max_length": 268.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 254.625, "completions/min_terminated_length": 233.0, "completions/max_terminated_length": 268.0, "rewards/meter/mean": 0.6067160367965698, "rewards/meter/std": 0.3461027145385742, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5166666507720947, "rewards/repeat_penalty/std": 0.17728106677532196, "rewards/total_composite/mean": 0.3126515746116638, "rewards/total_composite/std": 0.2052178531885147, "reward": 0.3126515746116638, "reward_std": 0.2052178531885147, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020115505903959274, "sampling/sampling_logp_difference/max": 1.3195867538452148, "sampling/importance_sampling_ratio/min": 0.26724570989608765, "sampling/importance_sampling_ratio/mean": 1.0007023811340332, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1408333508297801, "clip_ratio/low_mean": 0.009092815220355988, "clip_ratio/low_min": 0.009092815220355988, "clip_ratio/high_mean": 0.006283877417445183, "clip_ratio/high_max": 0.006283877417445183, "clip_ratio/region_mean": 0.01537669263780117, "reward_total_mean": 0.3126515746116638, "reward_meter_mean": 0.6067160367965698, "reward_meter_std": 0.3461027145385742, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5166666507720947, "reward_repeat_penalty_std": 0.17728106677532196, "reward_total_composite_mean": 0.3126515746116638, "reward_total_composite_std": 0.2052178531885147} {"timestamp_utc": "2026-04-11T23:05:25Z", "mode": "train", "global_step": 859, "epoch": 0.034502148853275495, "loss": 0.0073, "grad_norm": 1.447547435760498, "learning_rate": 7.4e-06, "num_tokens": 1916992.0, "completions/mean_length": 136.5, "completions/min_length": 129.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 136.5, "completions/min_terminated_length": 129.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.9829227924346924, "rewards/meter/std": 0.003464686218649149, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5535714626312256, "rewards/repeat_penalty/std": 0.23458294570446014, "rewards/total_composite/mean": 0.5436913371086121, "rewards/total_composite/std": 0.22969432175159454, "reward": 0.5436913371086121, "reward_std": 0.22969430685043335, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012742106802761555, "sampling/sampling_logp_difference/max": 1.1751036643981934, "sampling/importance_sampling_ratio/min": 0.30878695845603943, "sampling/importance_sampling_ratio/mean": 1.0010943412780762, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.048957847990095615, "clip_ratio/low_mean": 0.003665413591079414, "clip_ratio/low_min": 0.003665413591079414, "clip_ratio/high_mean": 0.008425170031841844, "clip_ratio/high_max": 0.008425170031841844, "clip_ratio/region_mean": 0.012090583622921258, "reward_total_mean": 0.5436913371086121, "reward_meter_mean": 0.9829227924346924, "reward_meter_std": 0.003464686218649149, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5535714626312256, "reward_repeat_penalty_std": 0.23458294570446014, "reward_total_composite_mean": 0.5436913371086121, "reward_total_composite_std": 0.22969432175159454} {"timestamp_utc": "2026-04-11T23:05:31Z", "mode": "train", "global_step": 860, "epoch": 0.03454231433506045, "loss": 0.0014, "grad_norm": 1.1458154916763306, "learning_rate": 7.396969696969698e-06, "num_tokens": 1919482.0, "completions/mean_length": 128.25, "completions/min_length": 122.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.25, "completions/min_terminated_length": 122.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.9833224415779114, "rewards/meter/std": 0.010524172335863113, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6428571939468384, "rewards/repeat_penalty/std": 0.15272071957588196, "rewards/total_composite/mean": 0.6316357851028442, "rewards/total_composite/std": 0.14875362813472748, "reward": 0.6316357851028442, "reward_std": 0.14875362813472748, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011860202066600323, "sampling/sampling_logp_difference/max": 1.2248728275299072, "sampling/importance_sampling_ratio/min": 0.29379504919052124, "sampling/importance_sampling_ratio/mean": 1.0009667873382568, "sampling/importance_sampling_ratio/max": 1.9968863725662231, "entropy": 0.03678457555361092, "clip_ratio/low_mean": 0.002871762844733894, "clip_ratio/low_min": 0.002871762844733894, "clip_ratio/high_mean": 0.005814186413772404, "clip_ratio/high_max": 0.005814186413772404, "clip_ratio/region_mean": 0.008685949258506298, "reward_total_mean": 0.6316357851028442, "reward_meter_mean": 0.9833224415779114, "reward_meter_std": 0.010524172335863113, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6428571939468384, "reward_repeat_penalty_std": 0.15272071957588196, "reward_total_composite_mean": 0.6316357851028442, "reward_total_composite_std": 0.14875362813472748} {"timestamp_utc": "2026-04-11T23:05:36Z", "mode": "train", "global_step": 861, "epoch": 0.0345824798168454, "loss": -0.0084, "grad_norm": 3.195672035217285, "learning_rate": 7.393939393939395e-06, "num_tokens": 1921968.0, "completions/mean_length": 130.75, "completions/min_length": 123.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 130.75, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.9875681400299072, "rewards/meter/std": 0.003578658914193511, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.660714328289032, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.6525848507881165, "rewards/total_composite/std": 0.07386582344770432, "reward": 0.6525848507881165, "reward_std": 0.07386582344770432, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009233459830284119, "sampling/sampling_logp_difference/max": 1.330092430114746, "sampling/importance_sampling_ratio/min": 0.26445281505584717, "sampling/importance_sampling_ratio/mean": 0.999982476234436, "sampling/importance_sampling_ratio/max": 1.8728368282318115, "entropy": 0.034579032799229026, "clip_ratio/low_mean": 0.002939337049610913, "clip_ratio/low_min": 0.002939337049610913, "clip_ratio/high_mean": 0.007541382568888366, "clip_ratio/high_max": 0.007541382568888366, "clip_ratio/region_mean": 0.010480719618499279, "reward_total_mean": 0.6525848507881165, "reward_meter_mean": 0.9875681400299072, "reward_meter_std": 0.003578658914193511, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.660714328289032, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.6525848507881165, "reward_total_composite_std": 0.07386582344770432} {"timestamp_utc": "2026-04-11T23:05:42Z", "mode": "train", "global_step": 862, "epoch": 0.03462264529863036, "loss": -0.026, "grad_norm": 4.800323963165283, "learning_rate": 7.390909090909092e-06, "num_tokens": 1924096.0, "completions/mean_length": 137.0, "completions/min_length": 123.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 137.0, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.8122666478157043, "rewards/meter/std": 0.3459046185016632, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7142857313156128, "rewards/repeat_penalty/std": 0.1322600096464157, "rewards/total_composite/mean": 0.5928546190261841, "rewards/total_composite/std": 0.2950608432292938, "reward": 0.5928546190261841, "reward_std": 0.29506081342697144, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024774322286248207, "sampling/sampling_logp_difference/max": 1.3954243659973145, "sampling/importance_sampling_ratio/min": 0.2477279007434845, "sampling/importance_sampling_ratio/mean": 1.0024539232254028, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0997177753597498, "clip_ratio/low_mean": 0.0039816161734052, "clip_ratio/low_min": 0.0039816161734052, "clip_ratio/high_mean": 0.010707754408940673, "clip_ratio/high_max": 0.010707754408940673, "clip_ratio/region_mean": 0.014689370582345873, "reward_total_mean": 0.5928546190261841, "reward_meter_mean": 0.8122666478157043, "reward_meter_std": 0.3459046185016632, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7142857313156128, "reward_repeat_penalty_std": 0.1322600096464157, "reward_total_composite_mean": 0.5928546190261841, "reward_total_composite_std": 0.2950608432292938} {"timestamp_utc": "2026-04-11T23:05:48Z", "mode": "train", "global_step": 863, "epoch": 0.03466281078041531, "loss": 0.0474, "grad_norm": 3.5590970516204834, "learning_rate": 7.3878787878787885e-06, "num_tokens": 1926759.0, "completions/mean_length": 147.875, "completions/min_length": 131.0, "completions/max_length": 152.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 147.875, "completions/min_terminated_length": 131.0, "completions/max_terminated_length": 152.0, "rewards/meter/mean": 0.9950444102287292, "rewards/meter/std": 0.0004635074583347887, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6904761791229248, "rewards/repeat_penalty/std": 0.06734350323677063, "rewards/total_composite/mean": 0.6657319068908691, "rewards/total_composite/std": 0.006685574073344469, "reward": 0.6657319068908691, "reward_std": 0.00668557733297348, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01787300407886505, "sampling/sampling_logp_difference/max": 2.7439072132110596, "sampling/importance_sampling_ratio/min": 0.06431855261325836, "sampling/importance_sampling_ratio/mean": 0.9978508353233337, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.036845392314717174, "clip_ratio/low_mean": 0.009212537202984095, "clip_ratio/low_min": 0.009212537202984095, "clip_ratio/high_mean": 0.0028625954873859882, "clip_ratio/high_max": 0.0028625954873859882, "clip_ratio/region_mean": 0.012075132690370083, "reward_total_mean": 0.6657319068908691, "reward_meter_mean": 0.9950444102287292, "reward_meter_std": 0.0004635074583347887, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6904761791229248, "reward_repeat_penalty_std": 0.06734350323677063, "reward_total_composite_mean": 0.6657319068908691, "reward_total_composite_std": 0.006685574073344469} {"timestamp_utc": "2026-04-11T23:05:53Z", "mode": "train", "global_step": 864, "epoch": 0.034702976262200265, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.384848484848486e-06, "num_tokens": 1928559.0, "completions/mean_length": 76.0, "completions/min_length": 76.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.0, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9935055375099182, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9935055375099182, "rewards/total_composite/std": 0.0, "reward": 0.9935055375099182, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0010724789462983608, "sampling/sampling_logp_difference/max": 0.06501199305057526, "sampling/importance_sampling_ratio/min": 0.9370561838150024, "sampling/importance_sampling_ratio/mean": 1.0007295608520508, "sampling/importance_sampling_ratio/max": 1.0238041877746582, "entropy": 0.010270603699609637, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9935055375099182, "reward_meter_mean": 0.9935055375099182, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9935055375099182, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-11T23:05:58Z", "mode": "train", "global_step": 865, "epoch": 0.03474314174398522, "loss": -0.002, "grad_norm": 3.38623309135437, "learning_rate": 7.381818181818182e-06, "num_tokens": 1930412.0, "completions/mean_length": 74.625, "completions/min_length": 74.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.625, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.995256781578064, "rewards/meter/std": 0.0008067663875408471, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.995256781578064, "rewards/total_composite/std": 0.0008067663875408471, "reward": 0.995256781578064, "reward_std": 0.0008067605085670948, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016164077445864677, "sampling/sampling_logp_difference/max": 1.9147498607635498, "sampling/importance_sampling_ratio/min": 0.14737869799137115, "sampling/importance_sampling_ratio/mean": 0.9971731305122375, "sampling/importance_sampling_ratio/max": 1.6714001893997192, "entropy": 0.07608733791857958, "clip_ratio/low_mean": 0.0016891892300918698, "clip_ratio/low_min": 0.0016891892300918698, "clip_ratio/high_mean": 0.00829179841093719, "clip_ratio/high_max": 0.00829179841093719, "clip_ratio/region_mean": 0.00998098764102906, "reward_total_mean": 0.995256781578064, "reward_meter_mean": 0.995256781578064, "reward_meter_std": 0.0008067663875408471, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.995256781578064, "reward_total_composite_std": 0.0008067663875408471} {"timestamp_utc": "2026-04-11T23:06:03Z", "mode": "train", "global_step": 866, "epoch": 0.03478330722577017, "loss": 0.1348, "grad_norm": 4.458425045013428, "learning_rate": 7.378787878787879e-06, "num_tokens": 1932315.0, "completions/mean_length": 79.875, "completions/min_length": 72.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.875, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.9195418953895569, "rewards/meter/std": 0.21018952131271362, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8895903825759888, "rewards/total_composite/std": 0.2949047088623047, "reward": 0.8895903825759888, "reward_std": 0.2949047088623047, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011505667120218277, "sampling/sampling_logp_difference/max": 1.7326593399047852, "sampling/importance_sampling_ratio/min": 0.17681357264518738, "sampling/importance_sampling_ratio/mean": 1.0000438690185547, "sampling/importance_sampling_ratio/max": 1.7624181509017944, "entropy": 0.029443061910569668, "clip_ratio/low_mean": 0.0022522523067891598, "clip_ratio/low_min": 0.0022522523067891598, "clip_ratio/high_mean": 0.004934210563078523, "clip_ratio/high_max": 0.004934210563078523, "clip_ratio/region_mean": 0.0071864628698676825, "reward_total_mean": 0.8895903825759888, "reward_meter_mean": 0.9195418953895569, "reward_meter_std": 0.21018952131271362, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.8895903825759888, "reward_total_composite_std": 0.2949047088623047} {"timestamp_utc": "2026-04-11T23:06:08Z", "mode": "train", "global_step": 867, "epoch": 0.03482347270755513, "loss": 0.0392, "grad_norm": 2.170355796813965, "learning_rate": 7.375757575757576e-06, "num_tokens": 1934484.0, "completions/mean_length": 98.125, "completions/min_length": 92.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.125, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.8766216039657593, "rewards/meter/std": 0.1744304597377777, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7012972831726074, "rewards/total_composite/std": 0.13954438269138336, "reward": 0.7012972831726074, "reward_std": 0.13954436779022217, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016189197078347206, "sampling/sampling_logp_difference/max": 2.60568904876709, "sampling/importance_sampling_ratio/min": 0.07385222613811493, "sampling/importance_sampling_ratio/mean": 1.000335454940796, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.05557233979925513, "clip_ratio/low_mean": 0.002314814832061529, "clip_ratio/low_min": 0.002314814832061529, "clip_ratio/high_mean": 0.010424387757666409, "clip_ratio/high_max": 0.010424387757666409, "clip_ratio/region_mean": 0.012739202589727938, "reward_total_mean": 0.7012972831726074, "reward_meter_mean": 0.8766216039657593, "reward_meter_std": 0.1744304597377777, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7012972831726074, "reward_total_composite_std": 0.13954438269138336} {"timestamp_utc": "2026-04-11T23:06:13Z", "mode": "train", "global_step": 868, "epoch": 0.03486363818934008, "loss": -0.0072, "grad_norm": 1.8654179573059082, "learning_rate": 7.372727272727274e-06, "num_tokens": 1936683.0, "completions/mean_length": 110.875, "completions/min_length": 107.0, "completions/max_length": 113.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 110.875, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 113.0, "rewards/meter/mean": 0.8900142908096313, "rewards/meter/std": 0.2951919138431549, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.712011456489563, "rewards/total_composite/std": 0.23615355789661407, "reward": 0.712011456489563, "reward_std": 0.23615355789661407, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004938235506415367, "sampling/sampling_logp_difference/max": 1.0195584297180176, "sampling/importance_sampling_ratio/min": 0.36075422167778015, "sampling/importance_sampling_ratio/mean": 1.0000348091125488, "sampling/importance_sampling_ratio/max": 1.3221184015274048, "entropy": 0.028168250457383692, "clip_ratio/low_mean": 0.004629629664123058, "clip_ratio/low_min": 0.004629629664123058, "clip_ratio/high_mean": 0.002336448524147272, "clip_ratio/high_max": 0.002336448524147272, "clip_ratio/region_mean": 0.00696607818827033, "reward_total_mean": 0.712011456489563, "reward_meter_mean": 0.8900142908096313, "reward_meter_std": 0.2951919138431549, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.712011456489563, "reward_total_composite_std": 0.23615355789661407} {"timestamp_utc": "2026-04-11T23:06:19Z", "mode": "train", "global_step": 869, "epoch": 0.034903803671125035, "loss": -0.019, "grad_norm": 3.4994313716888428, "learning_rate": 7.36969696969697e-06, "num_tokens": 1938765.0, "completions/mean_length": 107.25, "completions/min_length": 101.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.25, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.9882686138153076, "rewards/meter/std": 0.002476355992257595, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.8894296884536743, "rewards/total_composite/std": 0.10556221753358841, "reward": 0.8894296884536743, "reward_std": 0.10556221753358841, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02293246239423752, "sampling/sampling_logp_difference/max": 1.518688678741455, "sampling/importance_sampling_ratio/min": 0.21899887919425964, "sampling/importance_sampling_ratio/mean": 1.004834771156311, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12705577816814184, "clip_ratio/low_mean": 0.008417388889938593, "clip_ratio/low_min": 0.008417388889938593, "clip_ratio/high_mean": 0.010045865317806602, "clip_ratio/high_max": 0.010045865317806602, "clip_ratio/region_mean": 0.018463254207745194, "reward_total_mean": 0.8894296884536743, "reward_meter_mean": 0.9882686138153076, "reward_meter_std": 0.002476355992257595, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_total_composite_mean": 0.8894296884536743, "reward_total_composite_std": 0.10556221753358841} {"timestamp_utc": "2026-04-11T23:06:25Z", "mode": "train", "global_step": 870, "epoch": 0.03494396915290999, "loss": 0.0088, "grad_norm": 3.510559320449829, "learning_rate": 7.3666666666666676e-06, "num_tokens": 1941320.0, "completions/mean_length": 138.375, "completions/min_length": 133.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 138.375, "completions/min_terminated_length": 133.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.9983220100402832, "rewards/meter/std": 0.0005191374220885336, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8214285373687744, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.6560331583023071, "rewards/total_composite/std": 0.05272570997476578, "reward": 0.6560331583023071, "reward_std": 0.052725713700056076, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014895547181367874, "sampling/sampling_logp_difference/max": 1.4788875579833984, "sampling/importance_sampling_ratio/min": 0.22789107263088226, "sampling/importance_sampling_ratio/mean": 0.9968522191047668, "sampling/importance_sampling_ratio/max": 1.6303558349609375, "entropy": 0.05070830974727869, "clip_ratio/low_mean": 0.0017879948718473315, "clip_ratio/low_min": 0.0017879948718473315, "clip_ratio/high_mean": 0.015459735121112317, "clip_ratio/high_max": 0.015459735121112317, "clip_ratio/region_mean": 0.01724772999295965, "reward_total_mean": 0.6560331583023071, "reward_meter_mean": 0.9983220100402832, "reward_meter_std": 0.0005191374220885336, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8214285373687744, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.6560331583023071, "reward_total_composite_std": 0.05272570997476578} {"timestamp_utc": "2026-04-11T23:06:31Z", "mode": "train", "global_step": 871, "epoch": 0.03498413463469494, "loss": -0.0038, "grad_norm": 1.7928707599639893, "learning_rate": 7.363636363636364e-06, "num_tokens": 1944242.0, "completions/mean_length": 190.25, "completions/min_length": 175.0, "completions/max_length": 201.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 190.25, "completions/min_terminated_length": 175.0, "completions/max_terminated_length": 201.0, "rewards/meter/mean": 0.8820419907569885, "rewards/meter/std": 0.2552332878112793, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7638888955116272, "rewards/repeat_penalty/std": 0.0927247703075409, "rewards/total_composite/mean": 0.5602418184280396, "rewards/total_composite/std": 0.1772141009569168, "reward": 0.5602418184280396, "reward_std": 0.17721408605575562, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015695709735155106, "sampling/sampling_logp_difference/max": 1.124608039855957, "sampling/importance_sampling_ratio/min": 0.324779748916626, "sampling/importance_sampling_ratio/mean": 1.0031383037567139, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11951877502724528, "clip_ratio/low_mean": 0.0053914261516183615, "clip_ratio/low_min": 0.0053914261516183615, "clip_ratio/high_mean": 0.009079249284695834, "clip_ratio/high_max": 0.009079249284695834, "clip_ratio/region_mean": 0.014470675436314195, "reward_total_mean": 0.5602418184280396, "reward_meter_mean": 0.8820419907569885, "reward_meter_std": 0.2552332878112793, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7638888955116272, "reward_repeat_penalty_std": 0.0927247703075409, "reward_total_composite_mean": 0.5602418184280396, "reward_total_composite_std": 0.1772141009569168} {"timestamp_utc": "2026-04-11T23:06:37Z", "mode": "train", "global_step": 872, "epoch": 0.035024300116479896, "loss": -0.0316, "grad_norm": 3.245631694793701, "learning_rate": 7.360606060606061e-06, "num_tokens": 1946920.0, "completions/mean_length": 144.75, "completions/min_length": 129.0, "completions/max_length": 162.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 144.75, "completions/min_terminated_length": 129.0, "completions/max_terminated_length": 162.0, "rewards/meter/mean": 0.9848458170890808, "rewards/meter/std": 0.03508731722831726, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7250000238418579, "rewards/repeat_penalty/std": 0.030304575338959694, "rewards/total_composite/mean": 0.6906325221061707, "rewards/total_composite/std": 0.06134637072682381, "reward": 0.6906325221061707, "reward_std": 0.061346374452114105, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010910775512456894, "sampling/sampling_logp_difference/max": 1.9047117233276367, "sampling/importance_sampling_ratio/min": 0.14886555075645447, "sampling/importance_sampling_ratio/mean": 1.0042208433151245, "sampling/importance_sampling_ratio/max": 1.8771061897277832, "entropy": 0.05419332440942526, "clip_ratio/low_mean": 0.0038759689778089523, "clip_ratio/low_min": 0.0038759689778089523, "clip_ratio/high_mean": 0.005582162644714117, "clip_ratio/high_max": 0.005582162644714117, "clip_ratio/region_mean": 0.00945813162252307, "reward_total_mean": 0.6906325221061707, "reward_meter_mean": 0.9848458170890808, "reward_meter_std": 0.03508731722831726, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7250000238418579, "reward_repeat_penalty_std": 0.030304575338959694, "reward_total_composite_mean": 0.6906325221061707, "reward_total_composite_std": 0.06134637072682381} {"timestamp_utc": "2026-04-11T23:06:41Z", "mode": "train", "global_step": 873, "epoch": 0.03506446559826485, "loss": 0.0209, "grad_norm": 5.7755126953125, "learning_rate": 7.357575757575758e-06, "num_tokens": 1948679.0, "completions/mean_length": 64.875, "completions/min_length": 63.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.875, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9520989656448364, "rewards/meter/std": 0.055099111050367355, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9520989656448364, "rewards/total_composite/std": 0.055099111050367355, "reward": 0.9520989656448364, "reward_std": 0.05509910732507706, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.032992783933877945, "sampling/sampling_logp_difference/max": 1.5907907485961914, "sampling/importance_sampling_ratio/min": 0.2037644237279892, "sampling/importance_sampling_ratio/mean": 1.0100115537643433, "sampling/importance_sampling_ratio/max": 1.7566462755203247, "entropy": 0.17148890905082226, "clip_ratio/low_mean": 0.013034759555011988, "clip_ratio/low_min": 0.013034759555011988, "clip_ratio/high_mean": 0.027383167180232704, "clip_ratio/high_max": 0.027383167180232704, "clip_ratio/region_mean": 0.04041792673524469, "reward_total_mean": 0.9520989656448364, "reward_meter_mean": 0.9520989656448364, "reward_meter_std": 0.055099111050367355, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9520989656448364, "reward_total_composite_std": 0.055099111050367355} {"timestamp_utc": "2026-04-11T23:06:47Z", "mode": "train", "global_step": 874, "epoch": 0.035104631080049804, "loss": 0.0149, "grad_norm": 1.7160921096801758, "learning_rate": 7.354545454545456e-06, "num_tokens": 1950913.0, "completions/mean_length": 107.25, "completions/min_length": 105.0, "completions/max_length": 113.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.25, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 113.0, "rewards/meter/mean": 0.9955088496208191, "rewards/meter/std": 0.0011983213480561972, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7964070439338684, "rewards/total_composite/std": 0.0009586615487933159, "reward": 0.7964070439338684, "reward_std": 0.0009586560190655291, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008256379514932632, "sampling/sampling_logp_difference/max": 0.7744824886322021, "sampling/importance_sampling_ratio/min": 0.46094223856925964, "sampling/importance_sampling_ratio/mean": 0.9997270703315735, "sampling/importance_sampling_ratio/max": 1.3036043643951416, "entropy": 0.04265120159834623, "clip_ratio/low_mean": 0.0022123893722891808, "clip_ratio/low_min": 0.0022123893722891808, "clip_ratio/high_mean": 0.007142857299186289, "clip_ratio/high_max": 0.007142857299186289, "clip_ratio/region_mean": 0.00935524667147547, "reward_total_mean": 0.7964070439338684, "reward_meter_mean": 0.9955088496208191, "reward_meter_std": 0.0011983213480561972, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7964070439338684, "reward_total_composite_std": 0.0009586615487933159} {"timestamp_utc": "2026-04-11T23:06:52Z", "mode": "train", "global_step": 875, "epoch": 0.03514479656183476, "loss": -0.0005, "grad_norm": 2.395378351211548, "learning_rate": 7.351515151515151e-06, "num_tokens": 1953074.0, "completions/mean_length": 94.125, "completions/min_length": 87.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.125, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.992639422416687, "rewards/meter/std": 0.006307392846792936, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7749999761581421, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.7692322134971619, "rewards/total_composite/std": 0.06972821056842804, "reward": 0.7692322134971619, "reward_std": 0.06972822546958923, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02180575206875801, "sampling/sampling_logp_difference/max": 2.036391258239746, "sampling/importance_sampling_ratio/min": 0.13049879670143127, "sampling/importance_sampling_ratio/mean": 0.9954232573509216, "sampling/importance_sampling_ratio/max": 1.5037858486175537, "entropy": 0.07790858251973987, "clip_ratio/low_mean": 0.0026881720405071974, "clip_ratio/low_min": 0.0026881720405071974, "clip_ratio/high_mean": 0.023883325862698257, "clip_ratio/high_max": 0.023883325862698257, "clip_ratio/region_mean": 0.026571497903205454, "reward_total_mean": 0.7692322134971619, "reward_meter_mean": 0.992639422416687, "reward_meter_std": 0.006307392846792936, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7749999761581421, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.7692322134971619, "reward_total_composite_std": 0.06972821056842804} {"timestamp_utc": "2026-04-11T23:06:59Z", "mode": "train", "global_step": 876, "epoch": 0.03518496204361971, "loss": -0.0419, "grad_norm": 2.0017435550689697, "learning_rate": 7.348484848484849e-06, "num_tokens": 1956524.0, "completions/mean_length": 217.25, "completions/min_length": 206.0, "completions/max_length": 236.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 217.25, "completions/min_terminated_length": 206.0, "completions/max_terminated_length": 236.0, "rewards/meter/mean": 0.9868931770324707, "rewards/meter/std": 0.006763860583305359, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.0578637570142746, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6858974695205688, "rewards/repeat_penalty/std": 0.011869488283991814, "rewards/total_composite/mean": 0.6128624677658081, "rewards/total_composite/std": 0.02833697944879532, "reward": 0.6128624677658081, "reward_std": 0.02833697572350502, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012169601395726204, "sampling/sampling_logp_difference/max": 1.237346887588501, "sampling/importance_sampling_ratio/min": 0.29015299677848816, "sampling/importance_sampling_ratio/mean": 1.0016402006149292, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.052224091021344066, "clip_ratio/low_mean": 0.007715584768448025, "clip_ratio/low_min": 0.007715584768448025, "clip_ratio/high_mean": 0.0031779659911990166, "clip_ratio/high_max": 0.0031779659911990166, "clip_ratio/region_mean": 0.010893550759647042, "reward_total_mean": 0.6128624677658081, "reward_meter_mean": 0.9868931770324707, "reward_meter_std": 0.006763860583305359, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.0578637570142746, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6858974695205688, "reward_repeat_penalty_std": 0.011869488283991814, "reward_total_composite_mean": 0.6128624677658081, "reward_total_composite_std": 0.02833697944879532} {"timestamp_utc": "2026-04-11T23:07:04Z", "mode": "train", "global_step": 877, "epoch": 0.035225127525404666, "loss": -0.0079, "grad_norm": 6.4066290855407715, "learning_rate": 7.345454545454546e-06, "num_tokens": 1958416.0, "completions/mean_length": 79.5, "completions/min_length": 76.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.5, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.9876222610473633, "rewards/meter/std": 0.013733073137700558, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9876222610473633, "rewards/total_composite/std": 0.013733073137700558, "reward": 0.9876222610473633, "reward_std": 0.01373306754976511, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02791396901011467, "sampling/sampling_logp_difference/max": 1.3935667276382446, "sampling/importance_sampling_ratio/min": 0.24818849563598633, "sampling/importance_sampling_ratio/mean": 1.0038245916366577, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1384354429319501, "clip_ratio/low_mean": 0.003247300977818668, "clip_ratio/low_min": 0.003247300977818668, "clip_ratio/high_mean": 0.015677204821258783, "clip_ratio/high_max": 0.015677204821258783, "clip_ratio/region_mean": 0.01892450579907745, "reward_total_mean": 0.9876222610473633, "reward_meter_mean": 0.9876222610473633, "reward_meter_std": 0.013733073137700558, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9876222610473633, "reward_total_composite_std": 0.013733073137700558} {"timestamp_utc": "2026-04-11T23:07:09Z", "mode": "train", "global_step": 878, "epoch": 0.03526529300718962, "loss": -0.0205, "grad_norm": 3.125847816467285, "learning_rate": 7.342424242424243e-06, "num_tokens": 1960243.0, "completions/mean_length": 69.375, "completions/min_length": 64.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.375, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9980708360671997, "rewards/meter/std": 0.00046432489762082696, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980708360671997, "rewards/total_composite/std": 0.00046432489762082696, "reward": 0.9980708360671997, "reward_std": 0.00046432187082245946, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021711641922593117, "sampling/sampling_logp_difference/max": 0.9192183017730713, "sampling/importance_sampling_ratio/min": 0.39883068203926086, "sampling/importance_sampling_ratio/mean": 0.999176561832428, "sampling/importance_sampling_ratio/max": 1.6681042909622192, "entropy": 0.09143292810767889, "clip_ratio/low_mean": 0.016726079280488193, "clip_ratio/low_min": 0.016726079280488193, "clip_ratio/high_mean": 0.014086603187024593, "clip_ratio/high_max": 0.014086603187024593, "clip_ratio/region_mean": 0.030812682467512786, "reward_total_mean": 0.9980708360671997, "reward_meter_mean": 0.9980708360671997, "reward_meter_std": 0.00046432489762082696, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9980708360671997, "reward_total_composite_std": 0.00046432489762082696} {"timestamp_utc": "2026-04-11T23:07:14Z", "mode": "train", "global_step": 879, "epoch": 0.035305458488974574, "loss": -0.0466, "grad_norm": 4.135697841644287, "learning_rate": 7.3393939393939395e-06, "num_tokens": 1962012.0, "completions/mean_length": 65.125, "completions/min_length": 58.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.125, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.7186502814292908, "rewards/meter/std": 0.37441036105155945, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.7184880971908569, "rewards/total_composite/std": 0.3747643232345581, "reward": 0.7184880971908569, "reward_std": 0.37476426362991333, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025105303153395653, "sampling/sampling_logp_difference/max": 1.4957494735717773, "sampling/importance_sampling_ratio/min": 0.22408060729503632, "sampling/importance_sampling_ratio/mean": 0.9979657530784607, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.072424097917974, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.01865671609994024, "clip_ratio/high_max": 0.01865671609994024, "clip_ratio/region_mean": 0.01865671609994024, "reward_total_mean": 0.7184880971908569, "reward_meter_mean": 0.7186502814292908, "reward_meter_std": 0.37441036105155945, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.7184880971908569, "reward_total_composite_std": 0.3747643232345581} {"timestamp_utc": "2026-04-11T23:07:19Z", "mode": "train", "global_step": 880, "epoch": 0.03534562397075953, "loss": -0.0202, "grad_norm": 4.517680644989014, "learning_rate": 7.336363636363637e-06, "num_tokens": 1964149.0, "completions/mean_length": 94.125, "completions/min_length": 90.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.125, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.3598088026046753, "rewards/meter/std": 0.3960496783256531, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7749999761581421, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.28257015347480774, "rewards/total_composite/std": 0.31944534182548523, "reward": 0.28257015347480774, "reward_std": 0.31944531202316284, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015417142771184444, "sampling/sampling_logp_difference/max": 1.0558514595031738, "sampling/importance_sampling_ratio/min": 0.3478960692882538, "sampling/importance_sampling_ratio/mean": 0.9979188442230225, "sampling/importance_sampling_ratio/max": 1.8587523698806763, "entropy": 0.04804056999273598, "clip_ratio/low_mean": 0.008119725622236729, "clip_ratio/low_min": 0.008119725622236729, "clip_ratio/high_mean": 0.008080808212980628, "clip_ratio/high_max": 0.008080808212980628, "clip_ratio/region_mean": 0.016200533835217357, "reward_total_mean": 0.28257015347480774, "reward_meter_mean": 0.3598088026046753, "reward_meter_std": 0.3960496783256531, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7749999761581421, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.28257015347480774, "reward_total_composite_std": 0.31944534182548523} {"timestamp_utc": "2026-04-11T23:07:27Z", "mode": "train", "global_step": 881, "epoch": 0.03538578945254448, "loss": -0.0185, "grad_norm": 1.2314221858978271, "learning_rate": 7.333333333333333e-06, "num_tokens": 1968226.0, "completions/mean_length": 292.625, "completions/min_length": 277.0, "completions/max_length": 332.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 292.625, "completions/min_terminated_length": 277.0, "completions/max_terminated_length": 332.0, "rewards/meter/mean": 0.9959769248962402, "rewards/meter/std": 0.0035888429265469313, "rewards/count_adherence/mean": 0.7386363744735718, "rewards/count_adherence/std": 0.03214120864868164, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5985294580459595, "rewards/repeat_penalty/std": 0.004159451462328434, "rewards/total_composite/mean": 0.4401511549949646, "rewards/total_composite/std": 0.014130757190287113, "reward": 0.4401511549949646, "reward_std": 0.014130760915577412, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007294100243598223, "sampling/sampling_logp_difference/max": 0.9965987205505371, "sampling/importance_sampling_ratio/min": 0.3691328465938568, "sampling/importance_sampling_ratio/mean": 1.0001860857009888, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.04085008194670081, "clip_ratio/low_mean": 0.003381519520189613, "clip_ratio/low_min": 0.003381519520189613, "clip_ratio/high_mean": 0.002022653818130493, "clip_ratio/high_max": 0.002022653818130493, "clip_ratio/region_mean": 0.005404173338320106, "reward_total_mean": 0.4401511549949646, "reward_meter_mean": 0.9959769248962402, "reward_meter_std": 0.0035888429265469313, "reward_count_adherence_mean": 0.7386363744735718, "reward_count_adherence_std": 0.03214120864868164, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5985294580459595, "reward_repeat_penalty_std": 0.004159451462328434, "reward_total_composite_mean": 0.4401511549949646, "reward_total_composite_std": 0.014130757190287113} {"timestamp_utc": "2026-04-11T23:07:33Z", "mode": "train", "global_step": 882, "epoch": 0.035425954934329436, "loss": -0.002, "grad_norm": 3.3222203254699707, "learning_rate": 7.330303030303031e-06, "num_tokens": 1970988.0, "completions/mean_length": 144.25, "completions/min_length": 120.0, "completions/max_length": 166.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 144.25, "completions/min_terminated_length": 120.0, "completions/max_terminated_length": 166.0, "rewards/meter/mean": 0.7575975060462952, "rewards/meter/std": 0.32098710536956787, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.783730149269104, "rewards/repeat_penalty/std": 0.05935349687933922, "rewards/total_composite/mean": 0.5672966837882996, "rewards/total_composite/std": 0.25382858514785767, "reward": 0.5672966837882996, "reward_std": 0.2538285553455353, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013289229944348335, "sampling/sampling_logp_difference/max": 0.987497091293335, "sampling/importance_sampling_ratio/min": 0.37250787019729614, "sampling/importance_sampling_ratio/mean": 0.9993391633033752, "sampling/importance_sampling_ratio/max": 1.9269946813583374, "entropy": 0.05786414351314306, "clip_ratio/low_mean": 0.0054536922834813595, "clip_ratio/low_min": 0.0054536922834813595, "clip_ratio/high_mean": 0.0039639262249693274, "clip_ratio/high_max": 0.0039639262249693274, "clip_ratio/region_mean": 0.009417618508450687, "reward_total_mean": 0.5672966837882996, "reward_meter_mean": 0.7575975060462952, "reward_meter_std": 0.32098710536956787, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.783730149269104, "reward_repeat_penalty_std": 0.05935349687933922, "reward_total_composite_mean": 0.5672966837882996, "reward_total_composite_std": 0.25382858514785767} {"timestamp_utc": "2026-04-11T23:07:37Z", "mode": "train", "global_step": 883, "epoch": 0.03546612041611439, "loss": 0.0339, "grad_norm": 23.914064407348633, "learning_rate": 7.3272727272727285e-06, "num_tokens": 1972789.0, "completions/mean_length": 61.125, "completions/min_length": 55.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.125, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9858500361442566, "rewards/meter/std": 0.013770055025815964, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9858500361442566, "rewards/total_composite/std": 0.013770055025815964, "reward": 0.9858500361442566, "reward_std": 0.013770047575235367, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.058846138417720795, "sampling/sampling_logp_difference/max": 3.47511887550354, "sampling/importance_sampling_ratio/min": 0.030958153307437897, "sampling/importance_sampling_ratio/mean": 1.0076637268066406, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1264942679554224, "clip_ratio/low_mean": 0.00797146384138614, "clip_ratio/low_min": 0.00797146384138614, "clip_ratio/high_mean": 0.02063214615918696, "clip_ratio/high_max": 0.02063214615918696, "clip_ratio/region_mean": 0.0286036100005731, "reward_total_mean": 0.9858500361442566, "reward_meter_mean": 0.9858500361442566, "reward_meter_std": 0.013770055025815964, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9858500361442566, "reward_total_composite_std": 0.013770055025815964} {"timestamp_utc": "2026-04-11T23:07:44Z", "mode": "train", "global_step": 884, "epoch": 0.035506285897899344, "loss": 0.0017, "grad_norm": 2.2200710773468018, "learning_rate": 7.324242424242425e-06, "num_tokens": 1975890.0, "completions/mean_length": 178.625, "completions/min_length": 172.0, "completions/max_length": 187.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 178.625, "completions/min_terminated_length": 172.0, "completions/max_terminated_length": 187.0, "rewards/meter/mean": 0.6974109411239624, "rewards/meter/std": 0.3949660360813141, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6805555820465088, "rewards/repeat_penalty/std": 0.03928370773792267, "rewards/total_composite/mean": 0.39888328313827515, "rewards/total_composite/std": 0.23110367357730865, "reward": 0.39888328313827515, "reward_std": 0.23110367357730865, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015560545958578587, "sampling/sampling_logp_difference/max": 2.096597194671631, "sampling/importance_sampling_ratio/min": 0.12287382781505585, "sampling/importance_sampling_ratio/mean": 0.9986303448677063, "sampling/importance_sampling_ratio/max": 1.8987165689468384, "entropy": 0.05528515903279185, "clip_ratio/low_mean": 0.003500614082440734, "clip_ratio/low_min": 0.003500614082440734, "clip_ratio/high_mean": 0.008322211273480207, "clip_ratio/high_max": 0.008322211273480207, "clip_ratio/region_mean": 0.01182282535592094, "reward_total_mean": 0.39888328313827515, "reward_meter_mean": 0.6974109411239624, "reward_meter_std": 0.3949660360813141, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6805555820465088, "reward_repeat_penalty_std": 0.03928370773792267, "reward_total_composite_mean": 0.39888328313827515, "reward_total_composite_std": 0.23110367357730865} {"timestamp_utc": "2026-04-11T23:07:49Z", "mode": "train", "global_step": 885, "epoch": 0.0355464513796843, "loss": 0.0102, "grad_norm": 8.853222846984863, "learning_rate": 7.321212121212122e-06, "num_tokens": 1977749.0, "completions/mean_length": 70.375, "completions/min_length": 65.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.375, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.778793215751648, "rewards/meter/std": 0.4060738682746887, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.778793215751648, "rewards/total_composite/std": 0.4060738682746887, "reward": 0.778793215751648, "reward_std": 0.4060738682746887, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028910191729664803, "sampling/sampling_logp_difference/max": 2.095900535583496, "sampling/importance_sampling_ratio/min": 0.12295946478843689, "sampling/importance_sampling_ratio/mean": 0.9921159744262695, "sampling/importance_sampling_ratio/max": 1.356745958328247, "entropy": 0.10360199213027954, "clip_ratio/low_mean": 0.007144315168261528, "clip_ratio/low_min": 0.007144315168261528, "clip_ratio/high_mean": 0.025362646207213402, "clip_ratio/high_max": 0.025362646207213402, "clip_ratio/region_mean": 0.03250696137547493, "reward_total_mean": 0.778793215751648, "reward_meter_mean": 0.778793215751648, "reward_meter_std": 0.4060738682746887, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.778793215751648, "reward_total_composite_std": 0.4060738682746887} {"timestamp_utc": "2026-04-11T23:07:56Z", "mode": "train", "global_step": 886, "epoch": 0.03558661686146925, "loss": 0.0196, "grad_norm": 2.6481733322143555, "learning_rate": 7.3181818181818186e-06, "num_tokens": 1982015.0, "completions/mean_length": 294.25, "completions/min_length": 291.0, "completions/max_length": 308.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 294.25, "completions/min_terminated_length": 291.0, "completions/max_terminated_length": 308.0, "rewards/meter/mean": 0.8633161783218384, "rewards/meter/std": 0.3173806965351105, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5424836874008179, "rewards/repeat_penalty/std": 0.1294051706790924, "rewards/total_composite/mean": 0.45384329557418823, "rewards/total_composite/std": 0.1770940124988556, "reward": 0.45384329557418823, "reward_std": 0.1770940124988556, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003619949799031019, "sampling/sampling_logp_difference/max": 0.655219316482544, "sampling/importance_sampling_ratio/min": 0.537617027759552, "sampling/importance_sampling_ratio/mean": 1.0008361339569092, "sampling/importance_sampling_ratio/max": 1.9255647659301758, "entropy": 0.01867874281015247, "clip_ratio/low_mean": 0.0012175324372947216, "clip_ratio/low_min": 0.0012175324372947216, "clip_ratio/high_mean": 0.002134879759978503, "clip_ratio/high_max": 0.002134879759978503, "clip_ratio/region_mean": 0.0033524121972732246, "reward_total_mean": 0.45384329557418823, "reward_meter_mean": 0.8633161783218384, "reward_meter_std": 0.3173806965351105, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5424836874008179, "reward_repeat_penalty_std": 0.1294051706790924, "reward_total_composite_mean": 0.45384329557418823, "reward_total_composite_std": 0.1770940124988556} {"timestamp_utc": "2026-04-11T23:08:01Z", "mode": "train", "global_step": 887, "epoch": 0.035626782343254206, "loss": -0.0139, "grad_norm": 2.381399393081665, "learning_rate": 7.315151515151516e-06, "num_tokens": 1984181.0, "completions/mean_length": 105.75, "completions/min_length": 102.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.75, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.9935408234596252, "rewards/meter/std": 0.003044598735868931, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.8444880843162537, "rewards/total_composite/std": 0.09176833927631378, "reward": 0.8444880843162537, "reward_std": 0.09176833927631378, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010191258043050766, "sampling/sampling_logp_difference/max": 0.7071137428283691, "sampling/importance_sampling_ratio/min": 0.4930652678012848, "sampling/importance_sampling_ratio/mean": 1.0018950700759888, "sampling/importance_sampling_ratio/max": 1.6447700262069702, "entropy": 0.06286561442539096, "clip_ratio/low_mean": 0.005965773249045014, "clip_ratio/low_min": 0.005965773249045014, "clip_ratio/high_mean": 0.0046083927154541016, "clip_ratio/high_max": 0.0046083927154541016, "clip_ratio/region_mean": 0.010574165964499116, "reward_total_mean": 0.8444880843162537, "reward_meter_mean": 0.9935408234596252, "reward_meter_std": 0.003044598735868931, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.8444880843162537, "reward_total_composite_std": 0.09176833927631378} {"timestamp_utc": "2026-04-11T23:08:06Z", "mode": "train", "global_step": 888, "epoch": 0.03566694782503916, "loss": 0.0134, "grad_norm": 5.54829740524292, "learning_rate": 7.312121212121212e-06, "num_tokens": 1986101.0, "completions/mean_length": 62.0, "completions/min_length": 60.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.0, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9698032140731812, "rewards/meter/std": 0.021262098103761673, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9698032140731812, "rewards/total_composite/std": 0.021262098103761673, "reward": 0.9698032140731812, "reward_std": 0.021262086927890778, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019981540739536285, "sampling/sampling_logp_difference/max": 0.9673044681549072, "sampling/importance_sampling_ratio/min": 0.38010627031326294, "sampling/importance_sampling_ratio/mean": 1.0010225772857666, "sampling/importance_sampling_ratio/max": 1.7581480741500854, "entropy": 0.08477870561182499, "clip_ratio/low_mean": 0.003969253972172737, "clip_ratio/low_min": 0.003969253972172737, "clip_ratio/high_mean": 0.006115591386333108, "clip_ratio/high_max": 0.006115591386333108, "clip_ratio/region_mean": 0.010084845358505845, "reward_total_mean": 0.9698032140731812, "reward_meter_mean": 0.9698032140731812, "reward_meter_std": 0.021262098103761673, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9698032140731812, "reward_total_composite_std": 0.021262098103761673} {"timestamp_utc": "2026-04-11T23:08:11Z", "mode": "train", "global_step": 889, "epoch": 0.03570711330682411, "loss": 0.0078, "grad_norm": 3.3663084506988525, "learning_rate": 7.30909090909091e-06, "num_tokens": 1987943.0, "completions/mean_length": 72.25, "completions/min_length": 69.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.25, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9931055307388306, "rewards/meter/std": 0.0025689646136015654, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9931055307388306, "rewards/total_composite/std": 0.0025689646136015654, "reward": 0.9931055307388306, "reward_std": 0.002568964147940278, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021964317187666893, "sampling/sampling_logp_difference/max": 1.340165615081787, "sampling/importance_sampling_ratio/min": 0.2618023157119751, "sampling/importance_sampling_ratio/mean": 1.0010665655136108, "sampling/importance_sampling_ratio/max": 1.9120845794677734, "entropy": 0.09350762516260147, "clip_ratio/low_mean": 0.010253431391902268, "clip_ratio/low_min": 0.010253431391902268, "clip_ratio/high_mean": 0.005093896761536598, "clip_ratio/high_max": 0.005093896761536598, "clip_ratio/region_mean": 0.015347328153438866, "reward_total_mean": 0.9931055307388306, "reward_meter_mean": 0.9931055307388306, "reward_meter_std": 0.0025689646136015654, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9931055307388306, "reward_total_composite_std": 0.0025689646136015654} {"timestamp_utc": "2026-04-11T23:08:17Z", "mode": "train", "global_step": 890, "epoch": 0.03574727878860907, "loss": -0.005, "grad_norm": 2.6016640663146973, "learning_rate": 7.306060606060607e-06, "num_tokens": 1990330.0, "completions/mean_length": 145.375, "completions/min_length": 134.0, "completions/max_length": 160.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 145.375, "completions/min_terminated_length": 134.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.9830442667007446, "rewards/meter/std": 0.01036095805466175, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.75, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.7374788522720337, "rewards/total_composite/std": 0.0681464672088623, "reward": 0.7374788522720337, "reward_std": 0.06814645975828171, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021481409668922424, "sampling/sampling_logp_difference/max": 1.0175132751464844, "sampling/importance_sampling_ratio/min": 0.361492782831192, "sampling/importance_sampling_ratio/mean": 1.0078076124191284, "sampling/importance_sampling_ratio/max": 1.7661346197128296, "entropy": 0.146129728294909, "clip_ratio/low_mean": 0.014987432223279029, "clip_ratio/low_min": 0.014987432223279029, "clip_ratio/high_mean": 0.002556844614446163, "clip_ratio/high_max": 0.002556844614446163, "clip_ratio/region_mean": 0.017544276837725192, "reward_total_mean": 0.7374788522720337, "reward_meter_mean": 0.9830442667007446, "reward_meter_std": 0.01036095805466175, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.75, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.7374788522720337, "reward_total_composite_std": 0.0681464672088623} {"timestamp_utc": "2026-04-11T23:08:22Z", "mode": "train", "global_step": 891, "epoch": 0.03578744427039402, "loss": 0.0093, "grad_norm": 2.81315279006958, "learning_rate": 7.303030303030304e-06, "num_tokens": 1992094.0, "completions/mean_length": 73.5, "completions/min_length": 71.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.5, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.8690831661224365, "rewards/meter/std": 0.3090604245662689, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8690831661224365, "rewards/total_composite/std": 0.3090604245662689, "reward": 0.8690831661224365, "reward_std": 0.30906039476394653, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03247151896357536, "sampling/sampling_logp_difference/max": 1.3969697952270508, "sampling/importance_sampling_ratio/min": 0.24734534323215485, "sampling/importance_sampling_ratio/mean": 1.0030601024627686, "sampling/importance_sampling_ratio/max": 1.6693646907806396, "entropy": 0.19349019043147564, "clip_ratio/low_mean": 0.0016666667070239782, "clip_ratio/low_min": 0.0016666667070239782, "clip_ratio/high_mean": 0.018530045170336962, "clip_ratio/high_max": 0.018530045170336962, "clip_ratio/region_mean": 0.02019671187736094, "reward_total_mean": 0.8690831661224365, "reward_meter_mean": 0.8690831661224365, "reward_meter_std": 0.3090604245662689, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8690831661224365, "reward_total_composite_std": 0.3090604245662689} {"timestamp_utc": "2026-04-11T23:08:27Z", "mode": "train", "global_step": 892, "epoch": 0.035827609752178975, "loss": 0.1432, "grad_norm": 4.598532199859619, "learning_rate": 7.3e-06, "num_tokens": 1994275.0, "completions/mean_length": 99.625, "completions/min_length": 72.0, "completions/max_length": 115.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.625, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 115.0, "rewards/meter/mean": 0.994989275932312, "rewards/meter/std": 0.0016277554677799344, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.2314550280570984, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8250000476837158, "rewards/repeat_penalty/std": 0.12817399203777313, "rewards/total_composite/mean": 0.5349372625350952, "rewards/total_composite/std": 0.2865643799304962, "reward": 0.5349372625350952, "reward_std": 0.2865643799304962, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02673969604074955, "sampling/sampling_logp_difference/max": 1.9094719886779785, "sampling/importance_sampling_ratio/min": 0.14815859496593475, "sampling/importance_sampling_ratio/mean": 1.0079816579818726, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1309506855905056, "clip_ratio/low_mean": 0.013621801743283868, "clip_ratio/low_min": 0.013621801743283868, "clip_ratio/high_mean": 0.003402777831070125, "clip_ratio/high_max": 0.003402777831070125, "clip_ratio/region_mean": 0.017024579574353993, "reward_total_mean": 0.5349372625350952, "reward_meter_mean": 0.994989275932312, "reward_meter_std": 0.0016277554677799344, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.2314550280570984, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8250000476837158, "reward_repeat_penalty_std": 0.12817399203777313, "reward_total_composite_mean": 0.5349372625350952, "reward_total_composite_std": 0.2865643799304962} {"timestamp_utc": "2026-04-11T23:08:33Z", "mode": "train", "global_step": 893, "epoch": 0.03586777523396393, "loss": -0.0066, "grad_norm": 3.3807826042175293, "learning_rate": 7.296969696969698e-06, "num_tokens": 1996973.0, "completions/mean_length": 146.25, "completions/min_length": 143.0, "completions/max_length": 151.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 146.25, "completions/min_terminated_length": 143.0, "completions/max_terminated_length": 151.0, "rewards/meter/mean": 0.024410676211118698, "rewards/meter/std": 0.046276357024908066, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.75, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.017994364723563194, "rewards/total_composite/std": 0.03303234279155731, "reward": 0.017994364723563194, "reward_std": 0.033032339066267014, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023801125586032867, "sampling/sampling_logp_difference/max": 1.7393649816513062, "sampling/importance_sampling_ratio/min": 0.17563189566135406, "sampling/importance_sampling_ratio/mean": 1.000627875328064, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.13604794908314943, "clip_ratio/low_mean": 0.006828531681094319, "clip_ratio/low_min": 0.006828531681094319, "clip_ratio/high_mean": 0.009196062572300434, "clip_ratio/high_max": 0.009196062572300434, "clip_ratio/region_mean": 0.016024594253394753, "reward_total_mean": 0.017994364723563194, "reward_meter_mean": 0.024410676211118698, "reward_meter_std": 0.046276357024908066, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.75, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.017994364723563194, "reward_total_composite_std": 0.03303234279155731} {"timestamp_utc": "2026-04-11T23:08:38Z", "mode": "train", "global_step": 894, "epoch": 0.03590794071574888, "loss": 0.0158, "grad_norm": 3.678330659866333, "learning_rate": 7.293939393939394e-06, "num_tokens": 1998826.0, "completions/mean_length": 64.625, "completions/min_length": 63.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.625, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.957672119140625, "rewards/meter/std": 0.010208838619291782, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.957672119140625, "rewards/total_composite/std": 0.010208838619291782, "reward": 0.957672119140625, "reward_std": 0.01020884234458208, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00946664996445179, "sampling/sampling_logp_difference/max": 1.2841720581054688, "sampling/importance_sampling_ratio/min": 0.2768797278404236, "sampling/importance_sampling_ratio/mean": 0.9999814033508301, "sampling/importance_sampling_ratio/max": 1.4367311000823975, "entropy": 0.04335960140451789, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.005890377098694444, "clip_ratio/high_max": 0.005890377098694444, "clip_ratio/region_mean": 0.005890377098694444, "reward_total_mean": 0.957672119140625, "reward_meter_mean": 0.957672119140625, "reward_meter_std": 0.010208838619291782, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.957672119140625, "reward_total_composite_std": 0.010208838619291782} {"timestamp_utc": "2026-04-11T23:08:43Z", "mode": "train", "global_step": 895, "epoch": 0.03594810619753384, "loss": -0.0048, "grad_norm": 3.20367169380188, "learning_rate": 7.290909090909092e-06, "num_tokens": 2000706.0, "completions/mean_length": 67.0, "completions/min_length": 66.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.6329219341278076, "rewards/meter/std": 0.4643419682979584, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6329219341278076, "rewards/total_composite/std": 0.4643419682979584, "reward": 0.6329219341278076, "reward_std": 0.4643419682979584, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015903156250715256, "sampling/sampling_logp_difference/max": 0.6511527895927429, "sampling/importance_sampling_ratio/min": 0.5214443206787109, "sampling/importance_sampling_ratio/mean": 1.006116509437561, "sampling/importance_sampling_ratio/max": 1.8657705783843994, "entropy": 0.07575948582962155, "clip_ratio/low_mean": 0.011278833262622356, "clip_ratio/low_min": 0.011278833262622356, "clip_ratio/high_mean": 0.012951546465046704, "clip_ratio/high_max": 0.012951546465046704, "clip_ratio/region_mean": 0.02423037972766906, "reward_total_mean": 0.6329219341278076, "reward_meter_mean": 0.6329219341278076, "reward_meter_std": 0.4643419682979584, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6329219341278076, "reward_total_composite_std": 0.4643419682979584} {"timestamp_utc": "2026-04-11T23:08:48Z", "mode": "train", "global_step": 896, "epoch": 0.03598827167931879, "loss": 0.0424, "grad_norm": 17.783985137939453, "learning_rate": 7.287878787878789e-06, "num_tokens": 2002459.0, "completions/mean_length": 63.125, "completions/min_length": 59.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.125, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.926885187625885, "rewards/meter/std": 0.12749086320400238, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.926885187625885, "rewards/total_composite/std": 0.12749086320400238, "reward": 0.926885187625885, "reward_std": 0.12749086320400238, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04450727999210358, "sampling/sampling_logp_difference/max": 1.5940380096435547, "sampling/importance_sampling_ratio/min": 0.20310382544994354, "sampling/importance_sampling_ratio/mean": 0.9990435838699341, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.24534680880606174, "clip_ratio/low_mean": 0.007575757801532745, "clip_ratio/low_min": 0.007575757801532745, "clip_ratio/high_mean": 0.03242240101099014, "clip_ratio/high_max": 0.03242240101099014, "clip_ratio/region_mean": 0.03999815881252289, "reward_total_mean": 0.926885187625885, "reward_meter_mean": 0.926885187625885, "reward_meter_std": 0.12749086320400238, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.926885187625885, "reward_total_composite_std": 0.12749086320400238} {"timestamp_utc": "2026-04-11T23:08:53Z", "mode": "train", "global_step": 897, "epoch": 0.036028437161103745, "loss": 0.0518, "grad_norm": 9.401634216308594, "learning_rate": 7.284848484848486e-06, "num_tokens": 2004481.0, "completions/mean_length": 75.75, "completions/min_length": 72.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.75, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.9812333583831787, "rewards/meter/std": 0.03400373458862305, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9812333583831787, "rewards/total_composite/std": 0.03400373458862305, "reward": 0.9812333583831787, "reward_std": 0.03400372713804245, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04621249809861183, "sampling/sampling_logp_difference/max": 1.1664955615997314, "sampling/importance_sampling_ratio/min": 0.31145650148391724, "sampling/importance_sampling_ratio/mean": 1.0046457052230835, "sampling/importance_sampling_ratio/max": 1.990276575088501, "entropy": 0.3472878560423851, "clip_ratio/low_mean": 0.004629629664123058, "clip_ratio/low_min": 0.004629629664123058, "clip_ratio/high_mean": 0.02354571979958564, "clip_ratio/high_max": 0.02354571979958564, "clip_ratio/region_mean": 0.0281753494637087, "reward_total_mean": 0.9812333583831787, "reward_meter_mean": 0.9812333583831787, "reward_meter_std": 0.03400373458862305, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9812333583831787, "reward_total_composite_std": 0.03400373458862305} {"timestamp_utc": "2026-04-11T23:08:57Z", "mode": "train", "global_step": 898, "epoch": 0.0360686026428887, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.281818181818182e-06, "num_tokens": 2006258.0, "completions/mean_length": 64.125, "completions/min_length": 63.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.125, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9789108037948608, "rewards/meter/std": 0.013038679957389832, "rewards/count_adherence/mean": 0.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.0, "rewards/total_composite/std": 0.0, "reward": 0.0, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.01310182549059391, "sampling/sampling_logp_difference/max": 1.4251065254211426, "sampling/importance_sampling_ratio/min": 0.2404828518629074, "sampling/importance_sampling_ratio/mean": 1.0009260177612305, "sampling/importance_sampling_ratio/max": 1.383298635482788, "entropy": 0.07330859638750553, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.0, "reward_meter_mean": 0.9789108037948608, "reward_meter_std": 0.013038679957389832, "reward_count_adherence_mean": 0.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.0, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-11T23:09:03Z", "mode": "train", "global_step": 899, "epoch": 0.03610876812467365, "loss": -0.0124, "grad_norm": 3.2133944034576416, "learning_rate": 7.2787878787878795e-06, "num_tokens": 2008460.0, "completions/mean_length": 111.25, "completions/min_length": 101.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 111.25, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.7485231161117554, "rewards/meter/std": 0.4493827819824219, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.6480754613876343, "rewards/total_composite/std": 0.3985969126224518, "reward": 0.6480754613876343, "reward_std": 0.3985969126224518, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021135006099939346, "sampling/sampling_logp_difference/max": 1.0010976791381836, "sampling/importance_sampling_ratio/min": 0.36747586727142334, "sampling/importance_sampling_ratio/mean": 1.003222942352295, "sampling/importance_sampling_ratio/max": 1.8294382095336914, "entropy": 0.1524664619937539, "clip_ratio/low_mean": 0.004611999727785587, "clip_ratio/low_min": 0.004611999727785587, "clip_ratio/high_mean": 0.010862766648642719, "clip_ratio/high_max": 0.010862766648642719, "clip_ratio/region_mean": 0.015474766376428306, "reward_total_mean": 0.6480754613876343, "reward_meter_mean": 0.7485231161117554, "reward_meter_std": 0.4493827819824219, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.6480754613876343, "reward_total_composite_std": 0.3985969126224518} {"timestamp_utc": "2026-04-11T23:09:07Z", "mode": "train", "global_step": 900, "epoch": 0.03614893360645861, "loss": 0.0111, "grad_norm": 6.62265157699585, "learning_rate": 7.275757575757576e-06, "num_tokens": 2010132.0, "completions/mean_length": 54.0, "completions/min_length": 49.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.47188466787338257, "rewards/meter/std": 0.3173547685146332, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.47188466787338257, "rewards/total_composite/std": 0.3173547685146332, "reward": 0.47188466787338257, "reward_std": 0.3173547685146332, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05849804729223251, "sampling/sampling_logp_difference/max": 1.0531864166259766, "sampling/importance_sampling_ratio/min": 0.34882447123527527, "sampling/importance_sampling_ratio/mean": 1.0136620998382568, "sampling/importance_sampling_ratio/max": 1.9393072128295898, "entropy": 0.4540216512978077, "clip_ratio/low_mean": 0.024088865146040916, "clip_ratio/low_min": 0.024088865146040916, "clip_ratio/high_mean": 0.02068706788122654, "clip_ratio/high_max": 0.02068706788122654, "clip_ratio/region_mean": 0.044775933027267456, "reward_total_mean": 0.47188466787338257, "reward_meter_mean": 0.47188466787338257, "reward_meter_std": 0.3173547685146332, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.47188466787338257, "reward_total_composite_std": 0.3173547685146332} {"timestamp_utc": "2026-04-11T23:10:24Z", "mode": "eval", "global_step": 900, "epoch": 0.03614893360645861, "eval_loss": NaN, "eval_runtime": 76.3688, "eval_samples_per_second": 1.362, "eval_steps_per_second": 0.17, "eval_num_tokens": 2010132.0, "eval_completions/mean_length": 220.43269230769232, "eval_completions/min_length": 64.38461538461539, "eval_completions/max_length": 402.61538461538464, "eval_completions/clipped_ratio": 0.028846153846153848, "eval_completions/mean_terminated_length": 212.01373877892127, "eval_completions/min_terminated_length": 64.38461538461539, "eval_completions/max_terminated_length": 370.38461538461536, "eval_rewards/meter/mean": 0.607175561097952, "eval_rewards/meter/std": 0.4337661495575538, "eval_rewards/count_adherence/mean": 0.8996222019195557, "eval_rewards/count_adherence/std": 0.13804738968610764, "eval_rewards/arabic_clean/mean": 0.9807692307692307, "eval_rewards/arabic_clean/std": 0.05439282839114849, "eval_rewards/repeat_penalty/mean": 0.6879472640844492, "eval_rewards/repeat_penalty/std": 0.20650707471829194, "eval_rewards/total_composite/mean": 0.3643972415190477, "eval_rewards/total_composite/std": 0.31932372657152325, "eval_reward": 0.3643972415190477, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.008556847639668446, "eval_sampling/sampling_logp_difference/max": 0.8173254269819993, "eval_sampling/importance_sampling_ratio/min": 0.4500999932105725, "eval_sampling/importance_sampling_ratio/mean": 1.002149930367103, "eval_sampling/importance_sampling_ratio/max": 1.312718914105342, "eval_entropy": 0.0762252899316641, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.3643972415190477, "eval_reward_meter_mean": 0.607175561097952, "eval_reward_meter_std": 0.4337661495575538, "eval_reward_count_adherence_mean": 0.8996222019195557, "eval_reward_count_adherence_std": 0.13804738968610764, "eval_reward_arabic_clean_mean": 0.9807692307692307, "eval_reward_arabic_clean_std": 0.05439282839114849, "eval_reward_repeat_penalty_mean": 0.6879472640844492, "eval_reward_repeat_penalty_std": 0.20650707471829194, "eval_reward_total_composite_mean": 0.3643972415190477, "eval_reward_total_composite_std": 0.31932372657152325} {"timestamp_utc": "2026-04-11T23:10:32Z", "mode": "train", "global_step": 901, "epoch": 0.03618909908824356, "loss": 0.0221, "grad_norm": 4.90191650390625, "learning_rate": 7.272727272727273e-06, "num_tokens": 2011886.0, "completions/mean_length": 65.25, "completions/min_length": 62.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.25, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.2965307831764221, "rewards/meter/std": 0.3661009669303894, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.2965307831764221, "rewards/total_composite/std": 0.3661009669303894, "reward": 0.2965307831764221, "reward_std": 0.366100937128067, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05390980467200279, "sampling/sampling_logp_difference/max": 1.267104148864746, "sampling/importance_sampling_ratio/min": 0.28164607286453247, "sampling/importance_sampling_ratio/mean": 1.0055922269821167, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39315336383879185, "clip_ratio/low_mean": 0.02774310251697898, "clip_ratio/low_min": 0.02774310251697898, "clip_ratio/high_mean": 0.00974025996401906, "clip_ratio/high_max": 0.00974025996401906, "clip_ratio/region_mean": 0.03748336248099804, "reward_total_mean": 0.2965307831764221, "reward_meter_mean": 0.2965307831764221, "reward_meter_std": 0.3661009669303894, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.2965307831764221, "reward_total_composite_std": 0.3661009669303894} {"timestamp_utc": "2026-04-11T23:10:38Z", "mode": "train", "global_step": 902, "epoch": 0.036229264570028515, "loss": 0.1103, "grad_norm": 2.6275370121002197, "learning_rate": 7.26969696969697e-06, "num_tokens": 2014158.0, "completions/mean_length": 108.0, "completions/min_length": 92.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.0, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.9951065182685852, "rewards/meter/std": 0.0021594627760350704, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.17817415297031403, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7571429014205933, "rewards/repeat_penalty/std": 0.04581620916724205, "rewards/total_composite/mean": 0.6347411870956421, "rewards/total_composite/std": 0.17126502096652985, "reward": 0.6347411870956421, "reward_std": 0.17126502096652985, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019786102697253227, "sampling/sampling_logp_difference/max": 0.9821138381958008, "sampling/importance_sampling_ratio/min": 0.37451860308647156, "sampling/importance_sampling_ratio/mean": 1.0035145282745361, "sampling/importance_sampling_ratio/max": 1.7781134843826294, "entropy": 0.10153948981314898, "clip_ratio/low_mean": 0.011442428571172059, "clip_ratio/low_min": 0.011442428571172059, "clip_ratio/high_mean": 0.016934875398874283, "clip_ratio/high_max": 0.016934875398874283, "clip_ratio/region_mean": 0.02837730397004634, "reward_total_mean": 0.6347411870956421, "reward_meter_mean": 0.9951065182685852, "reward_meter_std": 0.0021594627760350704, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.17817415297031403, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7571429014205933, "reward_repeat_penalty_std": 0.04581620916724205, "reward_total_composite_mean": 0.6347411870956421, "reward_total_composite_std": 0.17126502096652985} {"timestamp_utc": "2026-04-11T23:10:43Z", "mode": "train", "global_step": 903, "epoch": 0.03626943005181347, "loss": -0.0025, "grad_norm": 1.540696620941162, "learning_rate": 7.266666666666668e-06, "num_tokens": 2016340.0, "completions/mean_length": 103.75, "completions/min_length": 103.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.75, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.9977596998214722, "rewards/meter/std": 0.00013270843192003667, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7749999761581421, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.773260235786438, "rewards/total_composite/std": 0.07051651179790497, "reward": 0.773260235786438, "reward_std": 0.07051652669906616, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012240896932780743, "sampling/sampling_logp_difference/max": 1.134617805480957, "sampling/importance_sampling_ratio/min": 0.32154497504234314, "sampling/importance_sampling_ratio/mean": 1.0004560947418213, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.05680669890716672, "clip_ratio/low_mean": 0.0012135922443121672, "clip_ratio/low_min": 0.0012135922443121672, "clip_ratio/high_mean": 0.009641934651881456, "clip_ratio/high_max": 0.009641934651881456, "clip_ratio/region_mean": 0.010855526896193624, "reward_total_mean": 0.773260235786438, "reward_meter_mean": 0.9977596998214722, "reward_meter_std": 0.00013270843192003667, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7749999761581421, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.773260235786438, "reward_total_composite_std": 0.07051651179790497} {"timestamp_utc": "2026-04-11T23:10:53Z", "mode": "train", "global_step": 904, "epoch": 0.03630959553359842, "loss": -0.0841, "grad_norm": 2.963177442550659, "learning_rate": 7.263636363636364e-06, "num_tokens": 2017997.0, "completions/mean_length": 97.125, "completions/min_length": 35.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 37.85714340209961, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.8145281076431274, "rewards/meter/std": 0.27075159549713135, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7298738360404968, "rewards/total_composite/std": 0.3964882493019104, "reward": 0.7298738360404968, "reward_std": 0.3964882493019104, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09322468191385269, "sampling/sampling_logp_difference/max": 1.8810420036315918, "sampling/importance_sampling_ratio/min": 0.1524311900138855, "sampling/importance_sampling_ratio/mean": 1.0168981552124023, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5050923340022564, "clip_ratio/low_mean": 0.00657894741743803, "clip_ratio/low_min": 0.00657894741743803, "clip_ratio/high_mean": 0.02894315170124173, "clip_ratio/high_max": 0.02894315170124173, "clip_ratio/region_mean": 0.03552209911867976, "reward_total_mean": 0.7298738360404968, "reward_meter_mean": 0.8145281076431274, "reward_meter_std": 0.27075159549713135, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7298738360404968, "reward_total_composite_std": 0.3964882493019104} {"timestamp_utc": "2026-04-11T23:10:59Z", "mode": "train", "global_step": 905, "epoch": 0.036349761015383376, "loss": 0.0081, "grad_norm": 1.4031760692596436, "learning_rate": 7.260606060606061e-06, "num_tokens": 2020527.0, "completions/mean_length": 153.25, "completions/min_length": 139.0, "completions/max_length": 162.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 153.25, "completions/min_terminated_length": 139.0, "completions/max_terminated_length": 162.0, "rewards/meter/mean": 0.9857839941978455, "rewards/meter/std": 0.010634319856762886, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6071429252624512, "rewards/repeat_penalty/std": 0.147871196269989, "rewards/total_composite/mean": 0.5992619395256042, "rewards/total_composite/std": 0.14893361926078796, "reward": 0.5992619395256042, "reward_std": 0.14893361926078796, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008156189695000648, "sampling/sampling_logp_difference/max": 0.697784423828125, "sampling/importance_sampling_ratio/min": 0.5704922080039978, "sampling/importance_sampling_ratio/mean": 1.0026439428329468, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.048865064047276974, "clip_ratio/low_mean": 0.00478401588043198, "clip_ratio/low_min": 0.00478401588043198, "clip_ratio/high_mean": 0.004120342840906233, "clip_ratio/high_max": 0.004120342840906233, "clip_ratio/region_mean": 0.008904358721338212, "reward_total_mean": 0.5992619395256042, "reward_meter_mean": 0.9857839941978455, "reward_meter_std": 0.010634319856762886, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6071429252624512, "reward_repeat_penalty_std": 0.147871196269989, "reward_total_composite_mean": 0.5992619395256042, "reward_total_composite_std": 0.14893361926078796} {"timestamp_utc": "2026-04-11T23:11:08Z", "mode": "train", "global_step": 906, "epoch": 0.03638992649716833, "loss": -0.0066, "grad_norm": 3.238330125808716, "learning_rate": 7.257575757575758e-06, "num_tokens": 2022039.0, "completions/mean_length": 98.0, "completions/min_length": 33.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 38.85714340209961, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.8625515699386597, "rewards/meter/std": 0.34917646646499634, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.4629100561141968, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7457597255706787, "rewards/total_composite/std": 0.4603087902069092, "reward": 0.7457597255706787, "reward_std": 0.4603087604045868, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.053011078387498856, "sampling/sampling_logp_difference/max": 2.6139426231384277, "sampling/importance_sampling_ratio/min": 0.07324519753456116, "sampling/importance_sampling_ratio/mean": 0.988756537437439, "sampling/importance_sampling_ratio/max": 1.4525083303451538, "entropy": 0.1836453713476658, "clip_ratio/low_mean": 0.003846153849735856, "clip_ratio/low_min": 0.003846153849735856, "clip_ratio/high_mean": 0.03257575840689242, "clip_ratio/high_max": 0.03257575840689242, "clip_ratio/region_mean": 0.036421912256628275, "reward_total_mean": 0.7457597255706787, "reward_meter_mean": 0.8625515699386597, "reward_meter_std": 0.34917646646499634, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.4629100561141968, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7457597255706787, "reward_total_composite_std": 0.4603087902069092} {"timestamp_utc": "2026-04-11T23:11:16Z", "mode": "train", "global_step": 907, "epoch": 0.036430091978953284, "loss": -0.0466, "grad_norm": 2.8838396072387695, "learning_rate": 7.254545454545455e-06, "num_tokens": 2025618.0, "completions/mean_length": 257.375, "completions/min_length": 241.0, "completions/max_length": 293.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 257.375, "completions/min_terminated_length": 241.0, "completions/max_terminated_length": 293.0, "rewards/meter/mean": 0.1675853729248047, "rewards/meter/std": 0.09037812799215317, "rewards/count_adherence/mean": 0.9166666269302368, "rewards/count_adherence/std": 0.05143444612622261, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.31433823704719543, "rewards/repeat_penalty/std": 0.18327650427818298, "rewards/total_composite/mean": 0.058102987706661224, "rewards/total_composite/std": 0.07150173932313919, "reward": 0.058102987706661224, "reward_std": 0.0715017318725586, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008167757652699947, "sampling/sampling_logp_difference/max": 1.5336222648620605, "sampling/importance_sampling_ratio/min": 0.2157527506351471, "sampling/importance_sampling_ratio/mean": 1.0021930932998657, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.030403183307498693, "clip_ratio/low_mean": 0.0029962139669805765, "clip_ratio/low_min": 0.0029962139669805765, "clip_ratio/high_mean": 0.0027064846362918615, "clip_ratio/high_max": 0.0027064846362918615, "clip_ratio/region_mean": 0.005702698603272438, "reward_total_mean": 0.058102987706661224, "reward_meter_mean": 0.1675853729248047, "reward_meter_std": 0.09037812799215317, "reward_count_adherence_mean": 0.9166666269302368, "reward_count_adherence_std": 0.05143444612622261, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.31433823704719543, "reward_repeat_penalty_std": 0.18327650427818298, "reward_total_composite_mean": 0.058102987706661224, "reward_total_composite_std": 0.07150173932313919} {"timestamp_utc": "2026-04-11T23:11:26Z", "mode": "train", "global_step": 908, "epoch": 0.03647025746073824, "loss": -0.183, "grad_norm": 0.40971025824546814, "learning_rate": 7.251515151515151e-06, "num_tokens": 2027442.0, "completions/mean_length": 132.0, "completions/min_length": 76.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 77.71428680419922, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.8670340180397034, "rewards/meter/std": 0.3502245545387268, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8669998645782471, "rewards/total_composite/std": 0.35032105445861816, "reward": 0.8669998645782471, "reward_std": 0.35032105445861816, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010826216079294682, "sampling/sampling_logp_difference/max": 0.651118278503418, "sampling/importance_sampling_ratio/min": 0.5214623212814331, "sampling/importance_sampling_ratio/mean": 1.0027331113815308, "sampling/importance_sampling_ratio/max": 1.2967445850372314, "entropy": 0.07274728454649448, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.007976973778568208, "clip_ratio/high_max": 0.007976973778568208, "clip_ratio/region_mean": 0.007976973778568208, "reward_total_mean": 0.8669998645782471, "reward_meter_mean": 0.8670340180397034, "reward_meter_std": 0.3502245545387268, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8669998645782471, "reward_total_composite_std": 0.35032105445861816} {"timestamp_utc": "2026-04-11T23:11:32Z", "mode": "train", "global_step": 909, "epoch": 0.03651042294252319, "loss": 0.0147, "grad_norm": 2.3503267765045166, "learning_rate": 7.2484848484848495e-06, "num_tokens": 2030075.0, "completions/mean_length": 151.125, "completions/min_length": 135.0, "completions/max_length": 158.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 151.125, "completions/min_terminated_length": 135.0, "completions/max_terminated_length": 158.0, "rewards/meter/mean": 0.8722712993621826, "rewards/meter/std": 0.33660420775413513, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.75, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.6414377689361572, "rewards/total_composite/std": 0.2504880726337433, "reward": 0.6414377689361572, "reward_std": 0.2504880726337433, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009546966291964054, "sampling/sampling_logp_difference/max": 0.9272332191467285, "sampling/importance_sampling_ratio/min": 0.395646870136261, "sampling/importance_sampling_ratio/mean": 1.0012683868408203, "sampling/importance_sampling_ratio/max": 1.8332045078277588, "entropy": 0.051733002765104175, "clip_ratio/low_mean": 0.003980891779065132, "clip_ratio/low_min": 0.003980891779065132, "clip_ratio/high_mean": 0.002550398523453623, "clip_ratio/high_max": 0.002550398523453623, "clip_ratio/region_mean": 0.006531290302518755, "reward_total_mean": 0.6414377689361572, "reward_meter_mean": 0.8722712993621826, "reward_meter_std": 0.33660420775413513, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.75, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.6414377689361572, "reward_total_composite_std": 0.2504880726337433} {"timestamp_utc": "2026-04-11T23:11:40Z", "mode": "train", "global_step": 910, "epoch": 0.03655058842430815, "loss": -0.0173, "grad_norm": 1.5391802787780762, "learning_rate": 7.245454545454546e-06, "num_tokens": 2034379.0, "completions/mean_length": 297.0, "completions/min_length": 274.0, "completions/max_length": 318.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 297.0, "completions/min_terminated_length": 274.0, "completions/max_terminated_length": 318.0, "rewards/meter/mean": 0.10647714138031006, "rewards/meter/std": 0.29925623536109924, "rewards/count_adherence/mean": 0.890625, "rewards/count_adherence/std": 0.08010874688625336, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6176573634147644, "rewards/repeat_penalty/std": 0.15113945305347443, "rewards/total_composite/mean": 0.06387612223625183, "rewards/total_composite/std": 0.1795579344034195, "reward": 0.06387612223625183, "reward_std": 0.1795579344034195, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02341293916106224, "sampling/sampling_logp_difference/max": 4.994693756103516, "sampling/importance_sampling_ratio/min": 0.0067737954668700695, "sampling/importance_sampling_ratio/mean": 0.999440610408783, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10688875103369355, "clip_ratio/low_mean": 0.011030830384697765, "clip_ratio/low_min": 0.011030830384697765, "clip_ratio/high_mean": 0.005175159312784672, "clip_ratio/high_max": 0.005175159312784672, "clip_ratio/region_mean": 0.016205989697482437, "reward_total_mean": 0.06387612223625183, "reward_meter_mean": 0.10647714138031006, "reward_meter_std": 0.29925623536109924, "reward_count_adherence_mean": 0.890625, "reward_count_adherence_std": 0.08010874688625336, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6176573634147644, "reward_repeat_penalty_std": 0.15113945305347443, "reward_total_composite_mean": 0.06387612223625183, "reward_total_composite_std": 0.1795579344034195} {"timestamp_utc": "2026-04-11T23:11:49Z", "mode": "train", "global_step": 911, "epoch": 0.03659075390609311, "loss": -0.0319, "grad_norm": 2.3471314907073975, "learning_rate": 7.242424242424243e-06, "num_tokens": 2036087.0, "completions/mean_length": 130.5, "completions/min_length": 72.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 76.0, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.00040913556586019695, "rewards/meter/std": 0.00048293921281583607, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.0003843327867798507, "rewards/total_composite/std": 0.0004887212999165058, "reward": 0.0003843327867798507, "reward_std": 0.0004887212999165058, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03155261650681496, "sampling/sampling_logp_difference/max": 1.4526575803756714, "sampling/importance_sampling_ratio/min": 0.2339477390050888, "sampling/importance_sampling_ratio/mean": 0.9987106919288635, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11146194022148848, "clip_ratio/low_mean": 0.012957202387042344, "clip_ratio/low_min": 0.012957202387042344, "clip_ratio/high_mean": 0.003086419776082039, "clip_ratio/high_max": 0.003086419776082039, "clip_ratio/region_mean": 0.016043622163124382, "reward_total_mean": 0.0003843327867798507, "reward_meter_mean": 0.00040913556586019695, "reward_meter_std": 0.00048293921281583607, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.0003843327867798507, "reward_total_composite_std": 0.0004887212999165058} {"timestamp_utc": "2026-04-11T23:11:59Z", "mode": "train", "global_step": 912, "epoch": 0.03663091938787806, "loss": -0.2509, "grad_norm": 0.514905571937561, "learning_rate": 7.2393939393939404e-06, "num_tokens": 2038908.0, "completions/mean_length": 227.625, "completions/min_length": 187.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 187.00001525878906, "completions/min_terminated_length": 187.0, "completions/max_terminated_length": 187.0, "rewards/meter/mean": 0.8691276907920837, "rewards/meter/std": 0.3413867950439453, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.7222222089767456, "rewards/repeat_penalty/std": 0.11878277361392975, "rewards/total_composite/mean": 0.5911375284194946, "rewards/total_composite/std": 0.24190568923950195, "reward": 0.5911375284194946, "reward_std": 0.24190570414066315, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0038684571627527475, "sampling/sampling_logp_difference/max": 0.8482571840286255, "sampling/importance_sampling_ratio/min": 0.42816048860549927, "sampling/importance_sampling_ratio/mean": 1.0010285377502441, "sampling/importance_sampling_ratio/max": 1.1373112201690674, "entropy": 0.026463798945769668, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.002673796727322042, "clip_ratio/high_max": 0.002673796727322042, "clip_ratio/region_mean": 0.002673796727322042, "reward_total_mean": 0.5911375284194946, "reward_meter_mean": 0.8691276907920837, "reward_meter_std": 0.3413867950439453, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.7222222089767456, "reward_repeat_penalty_std": 0.11878277361392975, "reward_total_composite_mean": 0.5911375284194946, "reward_total_composite_std": 0.24190568923950195} {"timestamp_utc": "2026-04-11T23:12:09Z", "mode": "train", "global_step": 913, "epoch": 0.036671084869663015, "loss": -0.0935, "grad_norm": 1.2060858011245728, "learning_rate": 7.236363636363637e-06, "num_tokens": 2042728.0, "completions/mean_length": 378.5, "completions/min_length": 328.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 334.0, "completions/min_terminated_length": 328.0, "completions/max_terminated_length": 343.0, "rewards/meter/mean": 0.24957457184791565, "rewards/meter/std": 0.42763951420783997, "rewards/count_adherence/mean": 0.6477272510528564, "rewards/count_adherence/std": 0.24021585285663605, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5911239385604858, "rewards/repeat_penalty/std": 0.21299511194229126, "rewards/total_composite/mean": 0.08265350759029388, "rewards/total_composite/std": 0.1630992591381073, "reward": 0.08265350759029388, "reward_std": 0.1630992740392685, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01384922955185175, "sampling/sampling_logp_difference/max": 2.1652774810791016, "sampling/importance_sampling_ratio/min": 0.11471810191869736, "sampling/importance_sampling_ratio/mean": 1.0017143487930298, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.04513176158070564, "clip_ratio/low_mean": 0.004096297547221184, "clip_ratio/low_min": 0.004096297547221184, "clip_ratio/high_mean": 0.0011353294248692691, "clip_ratio/high_max": 0.0011353294248692691, "clip_ratio/region_mean": 0.005231626972090453, "reward_total_mean": 0.08265350759029388, "reward_meter_mean": 0.24957457184791565, "reward_meter_std": 0.42763951420783997, "reward_count_adherence_mean": 0.6477272510528564, "reward_count_adherence_std": 0.24021585285663605, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5911239385604858, "reward_repeat_penalty_std": 0.21299511194229126, "reward_total_composite_mean": 0.08265350759029388, "reward_total_composite_std": 0.1630992591381073} {"timestamp_utc": "2026-04-11T23:12:14Z", "mode": "train", "global_step": 914, "epoch": 0.03671125035144797, "loss": 0.0031, "grad_norm": 3.0142698287963867, "learning_rate": 7.233333333333334e-06, "num_tokens": 2044450.0, "completions/mean_length": 69.25, "completions/min_length": 69.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.25, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9978233575820923, "rewards/meter/std": 0.000473748950753361, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978233575820923, "rewards/total_composite/std": 0.000473748950753361, "reward": 0.9978233575820923, "reward_std": 0.0004737446433864534, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017529845237731934, "sampling/sampling_logp_difference/max": 0.8328394889831543, "sampling/importance_sampling_ratio/min": 0.4556933641433716, "sampling/importance_sampling_ratio/mean": 1.0037966966629028, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09249463677406311, "clip_ratio/low_mean": 0.0053571430034935474, "clip_ratio/low_min": 0.0053571430034935474, "clip_ratio/high_mean": 0.009057971183210611, "clip_ratio/high_max": 0.009057971183210611, "clip_ratio/region_mean": 0.014415114186704159, "reward_total_mean": 0.9978233575820923, "reward_meter_mean": 0.9978233575820923, "reward_meter_std": 0.000473748950753361, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9978233575820923, "reward_total_composite_std": 0.000473748950753361} {"timestamp_utc": "2026-04-11T23:12:24Z", "mode": "train", "global_step": 915, "epoch": 0.03675141583323292, "loss": -0.2438, "grad_norm": 1.0550512075424194, "learning_rate": 7.2303030303030305e-06, "num_tokens": 2047404.0, "completions/mean_length": 410.25, "completions/min_length": 296.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.5, "completions/mean_terminated_length": 308.5, "completions/min_terminated_length": 296.0, "completions/max_terminated_length": 330.0, "rewards/meter/mean": 0.7131782174110413, "rewards/meter/std": 0.40048038959503174, "rewards/count_adherence/mean": 0.4027777910232544, "rewards/count_adherence/std": 0.36581191420555115, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.8327265977859497, "rewards/repeat_penalty/std": 0.20120510458946228, "rewards/total_composite/mean": 0.1930341273546219, "rewards/total_composite/std": 0.20283441245555878, "reward": 0.1930341273546219, "reward_std": 0.20283441245555878, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016473982483148575, "sampling/sampling_logp_difference/max": 2.0499348640441895, "sampling/importance_sampling_ratio/min": 0.12874329090118408, "sampling/importance_sampling_ratio/mean": 1.0016696453094482, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03112065652385354, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/high_mean": 0.00370671006385237, "clip_ratio/high_max": 0.00370671006385237, "clip_ratio/region_mean": 0.005600649514235556, "reward_total_mean": 0.1930341273546219, "reward_meter_mean": 0.7131782174110413, "reward_meter_std": 0.40048038959503174, "reward_count_adherence_mean": 0.4027777910232544, "reward_count_adherence_std": 0.36581191420555115, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.8327265977859497, "reward_repeat_penalty_std": 0.20120510458946228, "reward_total_composite_mean": 0.1930341273546219, "reward_total_composite_std": 0.20283441245555878} {"timestamp_utc": "2026-04-11T23:12:28Z", "mode": "train", "global_step": 916, "epoch": 0.03679158131501788, "loss": 0.1148, "grad_norm": 6.668519973754883, "learning_rate": 7.227272727272729e-06, "num_tokens": 2048990.0, "completions/mean_length": 54.25, "completions/min_length": 42.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.25, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.4157664477825165, "rewards/meter/std": 0.3722110092639923, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4157664477825165, "rewards/total_composite/std": 0.3722110092639923, "reward": 0.4157664477825165, "reward_std": 0.3722110092639923, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06838522106409073, "sampling/sampling_logp_difference/max": 2.055908203125, "sampling/importance_sampling_ratio/min": 0.12797655165195465, "sampling/importance_sampling_ratio/mean": 1.011838436126709, "sampling/importance_sampling_ratio/max": 1.8987327814102173, "entropy": 0.36291532032191753, "clip_ratio/low_mean": 0.02030206471681595, "clip_ratio/low_min": 0.02030206471681595, "clip_ratio/high_mean": 0.021643032785505056, "clip_ratio/high_max": 0.021643032785505056, "clip_ratio/region_mean": 0.041945097502321005, "reward_total_mean": 0.4157664477825165, "reward_meter_mean": 0.4157664477825165, "reward_meter_std": 0.3722110092639923, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.4157664477825165, "reward_total_composite_std": 0.3722110092639923} {"timestamp_utc": "2026-04-11T23:12:33Z", "mode": "train", "global_step": 917, "epoch": 0.03683174679680283, "loss": 0.0255, "grad_norm": 4.49209451675415, "learning_rate": 7.224242424242425e-06, "num_tokens": 2050857.0, "completions/mean_length": 81.375, "completions/min_length": 76.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.375, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.024357754737138748, "rewards/meter/std": 0.066401407122612, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.01962619088590145, "rewards/total_composite/std": 0.053068071603775024, "reward": 0.01962619088590145, "reward_std": 0.053068071603775024, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06128174066543579, "sampling/sampling_logp_difference/max": 2.628598213195801, "sampling/importance_sampling_ratio/min": 0.07217957079410553, "sampling/importance_sampling_ratio/mean": 1.002654790878296, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2575049940496683, "clip_ratio/low_mean": 0.05021517118439078, "clip_ratio/low_min": 0.05021517118439078, "clip_ratio/high_mean": 0.009615384973585606, "clip_ratio/high_max": 0.009615384973585606, "clip_ratio/region_mean": 0.05983055615797639, "reward_total_mean": 0.01962619088590145, "reward_meter_mean": 0.024357754737138748, "reward_meter_std": 0.066401407122612, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.01962619088590145, "reward_total_composite_std": 0.053068071603775024} {"timestamp_utc": "2026-04-11T23:12:38Z", "mode": "train", "global_step": 918, "epoch": 0.036871912278587785, "loss": 0.0014, "grad_norm": 3.6781203746795654, "learning_rate": 7.221212121212122e-06, "num_tokens": 2052361.0, "completions/mean_length": 40.0, "completions/min_length": 39.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.0, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.25423121452331543, "rewards/meter/std": 0.42384836077690125, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.25423121452331543, "rewards/total_composite/std": 0.42384836077690125, "reward": 0.25423121452331543, "reward_std": 0.42384836077690125, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025724250823259354, "sampling/sampling_logp_difference/max": 0.5552816390991211, "sampling/importance_sampling_ratio/min": 0.5739105939865112, "sampling/importance_sampling_ratio/mean": 1.0119025707244873, "sampling/importance_sampling_ratio/max": 1.5022114515304565, "entropy": 0.1840200126171112, "clip_ratio/low_mean": 0.015625000232830644, "clip_ratio/low_min": 0.015625000232830644, "clip_ratio/high_mean": 0.01266416534781456, "clip_ratio/high_max": 0.01266416534781456, "clip_ratio/region_mean": 0.028289165580645204, "reward_total_mean": 0.25423121452331543, "reward_meter_mean": 0.25423121452331543, "reward_meter_std": 0.42384836077690125, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.25423121452331543, "reward_total_composite_std": 0.42384836077690125} {"timestamp_utc": "2026-04-11T23:12:43Z", "mode": "train", "global_step": 919, "epoch": 0.03691207776037274, "loss": -0.0107, "grad_norm": 3.6577181816101074, "learning_rate": 7.218181818181819e-06, "num_tokens": 2054233.0, "completions/mean_length": 67.0, "completions/min_length": 64.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.8576065301895142, "rewards/meter/std": 0.346675843000412, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8576065301895142, "rewards/total_composite/std": 0.346675843000412, "reward": 0.8576065301895142, "reward_std": 0.3466758131980896, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024599751457571983, "sampling/sampling_logp_difference/max": 2.0653536319732666, "sampling/importance_sampling_ratio/min": 0.12677344679832458, "sampling/importance_sampling_ratio/mean": 1.0019348859786987, "sampling/importance_sampling_ratio/max": 1.7233705520629883, "entropy": 0.11905911657959223, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/high_mean": 0.023888979107141495, "clip_ratio/high_max": 0.023888979107141495, "clip_ratio/region_mean": 0.027795229107141495, "reward_total_mean": 0.8576065301895142, "reward_meter_mean": 0.8576065301895142, "reward_meter_std": 0.346675843000412, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8576065301895142, "reward_total_composite_std": 0.346675843000412} {"timestamp_utc": "2026-04-11T23:12:50Z", "mode": "train", "global_step": 920, "epoch": 0.03695224324215769, "loss": 0.0055, "grad_norm": 0.8415653109550476, "learning_rate": 7.215151515151516e-06, "num_tokens": 2058494.0, "completions/mean_length": 330.625, "completions/min_length": 313.0, "completions/max_length": 334.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 330.625, "completions/min_terminated_length": 313.0, "completions/max_terminated_length": 334.0, "rewards/meter/mean": 0.7762564420700073, "rewards/meter/std": 0.19550828635692596, "rewards/count_adherence/mean": 0.7692307829856873, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.625, "rewards/repeat_penalty/std": 0.0864252969622612, "rewards/total_composite/mean": 0.3763912320137024, "rewards/total_composite/std": 0.11739427596330643, "reward": 0.3763912320137024, "reward_std": 0.11739426851272583, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007077803369611502, "sampling/sampling_logp_difference/max": 0.9340815544128418, "sampling/importance_sampling_ratio/min": 0.39294660091400146, "sampling/importance_sampling_ratio/mean": 0.9997749328613281, "sampling/importance_sampling_ratio/max": 1.6070106029510498, "entropy": 0.03318166173994541, "clip_ratio/low_mean": 0.0014981299173086882, "clip_ratio/low_min": 0.0014981299173086882, "clip_ratio/high_mean": 0.005704862414859235, "clip_ratio/high_max": 0.005704862414859235, "clip_ratio/region_mean": 0.0072029923321679235, "reward_total_mean": 0.3763912320137024, "reward_meter_mean": 0.7762564420700073, "reward_meter_std": 0.19550828635692596, "reward_count_adherence_mean": 0.7692307829856873, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.625, "reward_repeat_penalty_std": 0.0864252969622612, "reward_total_composite_mean": 0.3763912320137024, "reward_total_composite_std": 0.11739427596330643} {"timestamp_utc": "2026-04-11T23:13:01Z", "mode": "train", "global_step": 921, "epoch": 0.036992408723942646, "loss": -0.1723, "grad_norm": 1.4356052875518799, "learning_rate": 7.212121212121212e-06, "num_tokens": 2060102.0, "completions/mean_length": 313.0, "completions/min_length": 108.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.5, "completions/mean_terminated_length": 114.0, "completions/min_terminated_length": 108.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.842616856098175, "rewards/meter/std": 0.3094305694103241, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.33034372329711914, "rewards/arabic_clean/mean": 0.5, "rewards/arabic_clean/std": 0.5345224738121033, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.44831955432891846, "rewards/total_composite/std": 0.4836740791797638, "reward": 0.44831955432891846, "reward_std": 0.4836740791797638, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0745995044708252, "sampling/sampling_logp_difference/max": 0.8662757873535156, "sampling/importance_sampling_ratio/min": 0.4205147325992584, "sampling/importance_sampling_ratio/mean": 1.0298396348953247, "sampling/importance_sampling_ratio/max": 1.9944467544555664, "entropy": 0.4247525855898857, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.029582271818071604, "clip_ratio/high_max": 0.029582271818071604, "clip_ratio/region_mean": 0.029582271818071604, "reward_total_mean": 0.44831955432891846, "reward_meter_mean": 0.842616856098175, "reward_meter_std": 0.3094305694103241, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.33034372329711914, "reward_arabic_clean_mean": 0.5, "reward_arabic_clean_std": 0.5345224738121033, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.44831955432891846, "reward_total_composite_std": 0.4836740791797638} {"timestamp_utc": "2026-04-11T23:13:08Z", "mode": "train", "global_step": 922, "epoch": 0.0370325742057276, "loss": -0.0231, "grad_norm": 1.704238772392273, "learning_rate": 7.2090909090909104e-06, "num_tokens": 2063555.0, "completions/mean_length": 248.625, "completions/min_length": 238.0, "completions/max_length": 266.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 248.625, "completions/min_terminated_length": 238.0, "completions/max_terminated_length": 266.0, "rewards/meter/mean": 0.007957300171256065, "rewards/meter/std": 0.0194723941385746, "rewards/count_adherence/mean": 0.8392857313156128, "rewards/count_adherence/std": 0.141575887799263, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6661838293075562, "rewards/repeat_penalty/std": 0.07680089771747589, "rewards/total_composite/mean": 0.004727967549115419, "rewards/total_composite/std": 0.012029902078211308, "reward": 0.004727967549115419, "reward_std": 0.012029902078211308, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02039029449224472, "sampling/sampling_logp_difference/max": 2.663630962371826, "sampling/importance_sampling_ratio/min": 0.06969470530748367, "sampling/importance_sampling_ratio/mean": 0.9993036985397339, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06539201783016324, "clip_ratio/low_mean": 0.012246111291460693, "clip_ratio/low_min": 0.012246111291460693, "clip_ratio/high_mean": 0.001409774413332343, "clip_ratio/high_max": 0.001409774413332343, "clip_ratio/region_mean": 0.013655885704793036, "reward_total_mean": 0.004727967549115419, "reward_meter_mean": 0.007957300171256065, "reward_meter_std": 0.0194723941385746, "reward_count_adherence_mean": 0.8392857313156128, "reward_count_adherence_std": 0.141575887799263, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6661838293075562, "reward_repeat_penalty_std": 0.07680089771747589, "reward_total_composite_mean": 0.004727967549115419, "reward_total_composite_std": 0.012029902078211308} {"timestamp_utc": "2026-04-11T23:13:13Z", "mode": "train", "global_step": 923, "epoch": 0.037072739687512554, "loss": -0.059, "grad_norm": 10.85976791381836, "learning_rate": 7.206060606060606e-06, "num_tokens": 2065157.0, "completions/mean_length": 49.25, "completions/min_length": 44.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.25, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.514995276927948, "rewards/meter/std": 0.3742930591106415, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.514995276927948, "rewards/total_composite/std": 0.3742930591106415, "reward": 0.514995276927948, "reward_std": 0.3742930591106415, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04450875520706177, "sampling/sampling_logp_difference/max": 1.193382740020752, "sampling/importance_sampling_ratio/min": 0.3031938970088959, "sampling/importance_sampling_ratio/mean": 1.0013222694396973, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17652267403900623, "clip_ratio/low_mean": 0.02332585956901312, "clip_ratio/low_min": 0.02332585956901312, "clip_ratio/high_mean": 0.027541207149624825, "clip_ratio/high_max": 0.027541207149624825, "clip_ratio/region_mean": 0.05086706671863794, "reward_total_mean": 0.514995276927948, "reward_meter_mean": 0.514995276927948, "reward_meter_std": 0.3742930591106415, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.514995276927948, "reward_total_composite_std": 0.3742930591106415} {"timestamp_utc": "2026-04-11T23:13:23Z", "mode": "train", "global_step": 924, "epoch": 0.03711290516929751, "loss": -0.127, "grad_norm": 1.6212074756622314, "learning_rate": 7.203030303030304e-06, "num_tokens": 2067180.0, "completions/mean_length": 214.875, "completions/min_length": 110.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 115.83333587646484, "completions/min_terminated_length": 110.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.594925582408905, "rewards/meter/std": 0.46331310272216797, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.40130022168159485, "rewards/total_composite/std": 0.3983116149902344, "reward": 0.40130022168159485, "reward_std": 0.39831164479255676, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.051524095237255096, "sampling/sampling_logp_difference/max": 1.1322174072265625, "sampling/importance_sampling_ratio/min": 0.3223177492618561, "sampling/importance_sampling_ratio/mean": 1.0083558559417725, "sampling/importance_sampling_ratio/max": 1.917328119277954, "entropy": 0.3687305059283972, "clip_ratio/low_mean": 0.01368181873112917, "clip_ratio/low_min": 0.01368181873112917, "clip_ratio/high_mean": 0.019490185426548123, "clip_ratio/high_max": 0.019490185426548123, "clip_ratio/region_mean": 0.03317200415767729, "reward_total_mean": 0.40130022168159485, "reward_meter_mean": 0.594925582408905, "reward_meter_std": 0.46331310272216797, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.40130022168159485, "reward_total_composite_std": 0.3983116149902344} {"timestamp_utc": "2026-04-11T23:13:28Z", "mode": "train", "global_step": 925, "epoch": 0.03715307065108246, "loss": -0.029, "grad_norm": 5.2026143074035645, "learning_rate": 7.2000000000000005e-06, "num_tokens": 2069136.0, "completions/mean_length": 73.5, "completions/min_length": 68.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.5, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.8316957354545593, "rewards/meter/std": 0.3097537159919739, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7086162567138672, "rewards/total_composite/std": 0.4172649383544922, "reward": 0.7086162567138672, "reward_std": 0.4172649383544922, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.051565829664468765, "sampling/sampling_logp_difference/max": 1.1075992584228516, "sampling/importance_sampling_ratio/min": 0.3303510844707489, "sampling/importance_sampling_ratio/mean": 1.0118038654327393, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36769964918494225, "clip_ratio/low_mean": 0.010874067898839712, "clip_ratio/low_min": 0.010874067898839712, "clip_ratio/high_mean": 0.03658821329008788, "clip_ratio/high_max": 0.03658821329008788, "clip_ratio/region_mean": 0.04746228118892759, "reward_total_mean": 0.7086162567138672, "reward_meter_mean": 0.8316957354545593, "reward_meter_std": 0.3097537159919739, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7086162567138672, "reward_total_composite_std": 0.4172649383544922} {"timestamp_utc": "2026-04-11T23:13:33Z", "mode": "train", "global_step": 926, "epoch": 0.037193236132867416, "loss": 0.001, "grad_norm": 2.8455419540405273, "learning_rate": 7.196969696969698e-06, "num_tokens": 2070956.0, "completions/mean_length": 69.5, "completions/min_length": 69.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.5, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9984892010688782, "rewards/meter/std": 8.65640613483265e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9984892010688782, "rewards/total_composite/std": 8.65640613483265e-05, "reward": 0.9984892010688782, "reward_std": 8.656938007334247e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015567532740533352, "sampling/sampling_logp_difference/max": 1.3253982067108154, "sampling/importance_sampling_ratio/min": 0.2656971514225006, "sampling/importance_sampling_ratio/mean": 1.0004963874816895, "sampling/importance_sampling_ratio/max": 1.7046222686767578, "entropy": 0.06283332780003548, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.00898033136036247, "clip_ratio/high_max": 0.00898033136036247, "clip_ratio/region_mean": 0.00898033136036247, "reward_total_mean": 0.9984892010688782, "reward_meter_mean": 0.9984892010688782, "reward_meter_std": 8.65640613483265e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9984892010688782, "reward_total_composite_std": 8.65640613483265e-05} {"timestamp_utc": "2026-04-11T23:13:42Z", "mode": "train", "global_step": 927, "epoch": 0.03723340161465237, "loss": -0.014, "grad_norm": 0.8467232584953308, "learning_rate": 7.193939393939394e-06, "num_tokens": 2075400.0, "completions/mean_length": 347.5, "completions/min_length": 341.0, "completions/max_length": 375.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 347.5, "completions/min_terminated_length": 341.0, "completions/max_terminated_length": 375.0, "rewards/meter/mean": 0.9979950189590454, "rewards/meter/std": 0.0002653496921993792, "rewards/count_adherence/mean": 0.7232142686843872, "rewards/count_adherence/std": 0.02525380253791809, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6365914344787598, "rewards/repeat_penalty/std": 0.01973436214029789, "rewards/total_composite/mean": 0.4593149423599243, "rewards/total_composite/std": 0.01691725291311741, "reward": 0.4593149423599243, "reward_std": 0.01691724918782711, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0067790052853524685, "sampling/sampling_logp_difference/max": 1.7181689739227295, "sampling/importance_sampling_ratio/min": 0.17939433455467224, "sampling/importance_sampling_ratio/mean": 1.0001847743988037, "sampling/importance_sampling_ratio/max": 1.8701292276382446, "entropy": 0.026407914701849222, "clip_ratio/low_mean": 0.00428855000063777, "clip_ratio/low_min": 0.00428855000063777, "clip_ratio/high_mean": 0.0010664711589924991, "clip_ratio/high_max": 0.0010664711589924991, "clip_ratio/region_mean": 0.005355021159630269, "reward_total_mean": 0.4593149423599243, "reward_meter_mean": 0.9979950189590454, "reward_meter_std": 0.0002653496921993792, "reward_count_adherence_mean": 0.7232142686843872, "reward_count_adherence_std": 0.02525380253791809, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6365914344787598, "reward_repeat_penalty_std": 0.01973436214029789, "reward_total_composite_mean": 0.4593149423599243, "reward_total_composite_std": 0.01691725291311741} {"timestamp_utc": "2026-04-11T23:13:49Z", "mode": "train", "global_step": 928, "epoch": 0.037273567096437324, "loss": 0.0067, "grad_norm": 1.9796561002731323, "learning_rate": 7.1909090909090914e-06, "num_tokens": 2078027.0, "completions/mean_length": 154.375, "completions/min_length": 150.0, "completions/max_length": 161.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 154.375, "completions/min_terminated_length": 150.0, "completions/max_terminated_length": 161.0, "rewards/meter/mean": 0.8792700171470642, "rewards/meter/std": 0.2660437226295471, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.645013689994812, "rewards/total_composite/std": 0.20092783868312836, "reward": 0.645013689994812, "reward_std": 0.20092782378196716, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013188260607421398, "sampling/sampling_logp_difference/max": 1.643843650817871, "sampling/importance_sampling_ratio/min": 0.19323588907718658, "sampling/importance_sampling_ratio/mean": 0.9983150959014893, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.04184746090322733, "clip_ratio/low_mean": 0.0015723269898444414, "clip_ratio/low_min": 0.0015723269898444414, "clip_ratio/high_mean": 0.008021060610190034, "clip_ratio/high_max": 0.008021060610190034, "clip_ratio/region_mean": 0.009593387600034475, "reward_total_mean": 0.645013689994812, "reward_meter_mean": 0.8792700171470642, "reward_meter_std": 0.2660437226295471, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.645013689994812, "reward_total_composite_std": 0.20092783868312836} {"timestamp_utc": "2026-04-11T23:13:55Z", "mode": "train", "global_step": 929, "epoch": 0.03731373257822228, "loss": 0.0038, "grad_norm": 2.6719958782196045, "learning_rate": 7.187878787878788e-06, "num_tokens": 2080954.0, "completions/mean_length": 170.875, "completions/min_length": 170.0, "completions/max_length": 171.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 170.875, "completions/min_terminated_length": 170.0, "completions/max_terminated_length": 171.0, "rewards/meter/mean": 0.9981188178062439, "rewards/meter/std": 0.0002052018535323441, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8055555820465088, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.8040474057197571, "rewards/total_composite/std": 0.051476798951625824, "reward": 0.8040474057197571, "reward_std": 0.05147679150104523, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008916568011045456, "sampling/sampling_logp_difference/max": 1.7634490728378296, "sampling/importance_sampling_ratio/min": 0.17145249247550964, "sampling/importance_sampling_ratio/mean": 0.9989221692085266, "sampling/importance_sampling_ratio/max": 1.3845103979110718, "entropy": 0.03398467996157706, "clip_ratio/low_mean": 0.003659270762000233, "clip_ratio/low_min": 0.003659270762000233, "clip_ratio/high_mean": 0.002923976629972458, "clip_ratio/high_max": 0.002923976629972458, "clip_ratio/region_mean": 0.006583247391972691, "reward_total_mean": 0.8040474057197571, "reward_meter_mean": 0.9981188178062439, "reward_meter_std": 0.0002052018535323441, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8055555820465088, "reward_repeat_penalty_std": 0.05143444985151291, "reward_total_composite_mean": 0.8040474057197571, "reward_total_composite_std": 0.051476798951625824} {"timestamp_utc": "2026-04-11T23:14:00Z", "mode": "train", "global_step": 930, "epoch": 0.03735389806000723, "loss": 0.0414, "grad_norm": 5.791079998016357, "learning_rate": 7.184848484848486e-06, "num_tokens": 2082718.0, "completions/mean_length": 77.5, "completions/min_length": 73.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.5, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.9255967140197754, "rewards/meter/std": 0.06689538806676865, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9255967140197754, "rewards/total_composite/std": 0.06689538806676865, "reward": 0.9255967140197754, "reward_std": 0.06689538061618805, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04519684612751007, "sampling/sampling_logp_difference/max": 1.1719346046447754, "sampling/importance_sampling_ratio/min": 0.30976709723472595, "sampling/importance_sampling_ratio/mean": 1.0131640434265137, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.26597489789128304, "clip_ratio/low_mean": 0.016923486720770597, "clip_ratio/low_min": 0.016923486720770597, "clip_ratio/high_mean": 0.016671410761773586, "clip_ratio/high_max": 0.016671410761773586, "clip_ratio/region_mean": 0.033594897482544184, "reward_total_mean": 0.9255967140197754, "reward_meter_mean": 0.9255967140197754, "reward_meter_std": 0.06689538806676865, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9255967140197754, "reward_total_composite_std": 0.06689538806676865} {"timestamp_utc": "2026-04-11T23:14:07Z", "mode": "train", "global_step": 931, "epoch": 0.037394063541792186, "loss": -0.0247, "grad_norm": 0.8847100734710693, "learning_rate": 7.181818181818182e-06, "num_tokens": 2086955.0, "completions/mean_length": 309.625, "completions/min_length": 289.0, "completions/max_length": 322.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 309.625, "completions/min_terminated_length": 289.0, "completions/max_terminated_length": 322.0, "rewards/meter/mean": 0.7983001470565796, "rewards/meter/std": 0.3538459241390228, "rewards/count_adherence/mean": 0.8888888955116272, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6083333492279053, "rewards/repeat_penalty/std": 0.0235702246427536, "rewards/total_composite/mean": 0.4331192076206207, "rewards/total_composite/std": 0.19441884756088257, "reward": 0.4331192076206207, "reward_std": 0.19441883265972137, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008012884296476841, "sampling/sampling_logp_difference/max": 1.079960823059082, "sampling/importance_sampling_ratio/min": 0.3396088182926178, "sampling/importance_sampling_ratio/mean": 0.9997681379318237, "sampling/importance_sampling_ratio/max": 1.7735785245895386, "entropy": 0.031651133904233575, "clip_ratio/low_mean": 0.00041946308920159936, "clip_ratio/low_min": 0.00041946308920159936, "clip_ratio/high_mean": 0.009181037603411824, "clip_ratio/high_max": 0.009181037603411824, "clip_ratio/region_mean": 0.009600500692613423, "reward_total_mean": 0.4331192076206207, "reward_meter_mean": 0.7983001470565796, "reward_meter_std": 0.3538459241390228, "reward_count_adherence_mean": 0.8888888955116272, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6083333492279053, "reward_repeat_penalty_std": 0.0235702246427536, "reward_total_composite_mean": 0.4331192076206207, "reward_total_composite_std": 0.19441884756088257} {"timestamp_utc": "2026-04-11T23:14:13Z", "mode": "train", "global_step": 932, "epoch": 0.03743422902357714, "loss": -0.0003, "grad_norm": 2.1663684844970703, "learning_rate": 7.17878787878788e-06, "num_tokens": 2089791.0, "completions/mean_length": 148.5, "completions/min_length": 146.0, "completions/max_length": 150.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 148.5, "completions/min_terminated_length": 146.0, "completions/max_terminated_length": 150.0, "rewards/meter/mean": 0.7308982610702515, "rewards/meter/std": 0.4457210898399353, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.5223252773284912, "rewards/total_composite/std": 0.317903995513916, "reward": 0.5223252773284912, "reward_std": 0.317903995513916, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012427828274667263, "sampling/sampling_logp_difference/max": 1.0588593482971191, "sampling/importance_sampling_ratio/min": 0.3468512296676636, "sampling/importance_sampling_ratio/mean": 0.9969629049301147, "sampling/importance_sampling_ratio/max": 1.8277068138122559, "entropy": 0.037923737429082394, "clip_ratio/low_mean": 0.0017123287543654442, "clip_ratio/low_min": 0.0017123287543654442, "clip_ratio/high_mean": 0.010114155360497534, "clip_ratio/high_max": 0.010114155360497534, "clip_ratio/region_mean": 0.011826484114862978, "reward_total_mean": 0.5223252773284912, "reward_meter_mean": 0.7308982610702515, "reward_meter_std": 0.4457210898399353, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.5223252773284912, "reward_total_composite_std": 0.317903995513916} {"timestamp_utc": "2026-04-11T23:14:18Z", "mode": "train", "global_step": 933, "epoch": 0.037474394505362094, "loss": -0.0034, "grad_norm": 8.870975494384766, "learning_rate": 7.175757575757576e-06, "num_tokens": 2091401.0, "completions/mean_length": 53.25, "completions/min_length": 52.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.25, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9159232378005981, "rewards/meter/std": 0.05563880503177643, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9159232378005981, "rewards/total_composite/std": 0.05563880503177643, "reward": 0.9159232378005981, "reward_std": 0.05563879758119583, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023871900513768196, "sampling/sampling_logp_difference/max": 1.3186838626861572, "sampling/importance_sampling_ratio/min": 0.26748713850975037, "sampling/importance_sampling_ratio/mean": 1.0049976110458374, "sampling/importance_sampling_ratio/max": 1.4143028259277344, "entropy": 0.1050838534720242, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.009302935097366571, "clip_ratio/high_max": 0.009302935097366571, "clip_ratio/region_mean": 0.009302935097366571, "reward_total_mean": 0.9159232378005981, "reward_meter_mean": 0.9159232378005981, "reward_meter_std": 0.05563880503177643, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9159232378005981, "reward_total_composite_std": 0.05563880503177643} {"timestamp_utc": "2026-04-11T23:14:23Z", "mode": "train", "global_step": 934, "epoch": 0.03751455998714705, "loss": 0.0017, "grad_norm": 6.214112281799316, "learning_rate": 7.172727272727273e-06, "num_tokens": 2093250.0, "completions/mean_length": 69.125, "completions/min_length": 66.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.125, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.40722712874412537, "rewards/meter/std": 0.28478607535362244, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.40722712874412537, "rewards/total_composite/std": 0.28478607535362244, "reward": 0.40722712874412537, "reward_std": 0.28478607535362244, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03458065912127495, "sampling/sampling_logp_difference/max": 1.8944740295410156, "sampling/importance_sampling_ratio/min": 0.1503974199295044, "sampling/importance_sampling_ratio/mean": 0.9984850883483887, "sampling/importance_sampling_ratio/max": 1.4551042318344116, "entropy": 0.1644162703305483, "clip_ratio/low_mean": 0.01640054234303534, "clip_ratio/low_min": 0.01640054234303534, "clip_ratio/high_mean": 0.016192146576941013, "clip_ratio/high_max": 0.016192146576941013, "clip_ratio/region_mean": 0.032592688919976354, "reward_total_mean": 0.40722712874412537, "reward_meter_mean": 0.40722712874412537, "reward_meter_std": 0.28478607535362244, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.40722712874412537, "reward_total_composite_std": 0.28478607535362244} {"timestamp_utc": "2026-04-11T23:14:27Z", "mode": "train", "global_step": 935, "epoch": 0.037554725468932, "loss": 0.0203, "grad_norm": 7.902743816375732, "learning_rate": 7.16969696969697e-06, "num_tokens": 2094727.0, "completions/mean_length": 32.625, "completions/min_length": 32.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.625, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9710537195205688, "rewards/meter/std": 0.003800545586273074, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9710537195205688, "rewards/total_composite/std": 0.003800545586273074, "reward": 0.9710537195205688, "reward_std": 0.003800547681748867, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009311332367360592, "sampling/sampling_logp_difference/max": 0.41676831245422363, "sampling/importance_sampling_ratio/min": 0.7209804058074951, "sampling/importance_sampling_ratio/mean": 1.0037425756454468, "sampling/importance_sampling_ratio/max": 1.517050862312317, "entropy": 0.051071187015622854, "clip_ratio/low_mean": 0.011363636702299118, "clip_ratio/low_min": 0.011363636702299118, "clip_ratio/high_mean": 0.00390625, "clip_ratio/high_max": 0.00390625, "clip_ratio/region_mean": 0.015269886702299118, "reward_total_mean": 0.9710537195205688, "reward_meter_mean": 0.9710537195205688, "reward_meter_std": 0.003800545586273074, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9710537195205688, "reward_total_composite_std": 0.003800545586273074} {"timestamp_utc": "2026-04-11T23:14:32Z", "mode": "train", "global_step": 936, "epoch": 0.037594890950716955, "loss": 0.008, "grad_norm": 12.27761173248291, "learning_rate": 7.166666666666667e-06, "num_tokens": 2096787.0, "completions/mean_length": 77.5, "completions/min_length": 76.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.5, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9931102991104126, "rewards/meter/std": 0.006781714037060738, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9931102991104126, "rewards/total_composite/std": 0.006781714037060738, "reward": 0.9931102991104126, "reward_std": 0.006781699601560831, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017138954252004623, "sampling/sampling_logp_difference/max": 0.852900505065918, "sampling/importance_sampling_ratio/min": 0.4686746895313263, "sampling/importance_sampling_ratio/mean": 1.001671314239502, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09636378474533558, "clip_ratio/low_mean": 0.0015625000232830644, "clip_ratio/low_min": 0.0015625000232830644, "clip_ratio/high_mean": 0.0048290317645296454, "clip_ratio/high_max": 0.0048290317645296454, "clip_ratio/region_mean": 0.00639153178781271, "reward_total_mean": 0.9931102991104126, "reward_meter_mean": 0.9931102991104126, "reward_meter_std": 0.006781714037060738, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9931102991104126, "reward_total_composite_std": 0.006781714037060738} {"timestamp_utc": "2026-04-11T23:14:37Z", "mode": "train", "global_step": 937, "epoch": 0.03763505643250191, "loss": 0.0106, "grad_norm": 9.313652992248535, "learning_rate": 7.163636363636363e-06, "num_tokens": 2098529.0, "completions/mean_length": 68.75, "completions/min_length": 67.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.75, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9961028099060059, "rewards/meter/std": 0.0017646643100306392, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9961028099060059, "rewards/total_composite/std": 0.0017646643100306392, "reward": 0.9961028099060059, "reward_std": 0.001764659769833088, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028187252581119537, "sampling/sampling_logp_difference/max": 1.3503470420837402, "sampling/importance_sampling_ratio/min": 0.25915029644966125, "sampling/importance_sampling_ratio/mean": 1.0019060373306274, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0943189668469131, "clip_ratio/low_mean": 0.0036231884732842445, "clip_ratio/low_min": 0.0036231884732842445, "clip_ratio/high_mean": 0.014654986094683409, "clip_ratio/high_max": 0.014654986094683409, "clip_ratio/region_mean": 0.018278174567967653, "reward_total_mean": 0.9961028099060059, "reward_meter_mean": 0.9961028099060059, "reward_meter_std": 0.0017646643100306392, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9961028099060059, "reward_total_composite_std": 0.0017646643100306392} {"timestamp_utc": "2026-04-11T23:14:43Z", "mode": "train", "global_step": 938, "epoch": 0.03767522191428686, "loss": -0.0077, "grad_norm": 2.60717511177063, "learning_rate": 7.1606060606060615e-06, "num_tokens": 2101075.0, "completions/mean_length": 150.25, "completions/min_length": 144.0, "completions/max_length": 155.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 150.25, "completions/min_terminated_length": 144.0, "completions/max_terminated_length": 155.0, "rewards/meter/mean": 0.9614237546920776, "rewards/meter/std": 0.02061135321855545, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7142857313156128, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.686731219291687, "rewards/total_composite/std": 0.014722409658133984, "reward": 0.686731219291687, "reward_std": 0.014722414314746857, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005851758643984795, "sampling/sampling_logp_difference/max": 1.237354040145874, "sampling/importance_sampling_ratio/min": 0.2901509404182434, "sampling/importance_sampling_ratio/mean": 1.0006721019744873, "sampling/importance_sampling_ratio/max": 1.5528877973556519, "entropy": 0.02918590500485152, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.004902798566035926, "clip_ratio/high_max": 0.004902798566035926, "clip_ratio/region_mean": 0.004902798566035926, "reward_total_mean": 0.686731219291687, "reward_meter_mean": 0.9614237546920776, "reward_meter_std": 0.02061135321855545, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7142857313156128, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.686731219291687, "reward_total_composite_std": 0.014722409658133984} {"timestamp_utc": "2026-04-11T23:14:47Z", "mode": "train", "global_step": 939, "epoch": 0.03771538739607182, "loss": -0.0037, "grad_norm": 12.418314933776855, "learning_rate": 7.157575757575758e-06, "num_tokens": 2102575.0, "completions/mean_length": 36.5, "completions/min_length": 34.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.5, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.21266832947731018, "rewards/meter/std": 0.20291058719158173, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.21266832947731018, "rewards/total_composite/std": 0.20291058719158173, "reward": 0.21266832947731018, "reward_std": 0.20291060209274292, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03699643164873123, "sampling/sampling_logp_difference/max": 1.311621904373169, "sampling/importance_sampling_ratio/min": 0.2693828046321869, "sampling/importance_sampling_ratio/mean": 1.0057456493377686, "sampling/importance_sampling_ratio/max": 1.8581043481826782, "entropy": 0.13651060685515404, "clip_ratio/low_mean": 0.017079579178243876, "clip_ratio/low_min": 0.017079579178243876, "clip_ratio/high_mean": 0.013518452877178788, "clip_ratio/high_max": 0.013518452877178788, "clip_ratio/region_mean": 0.030598032055422664, "reward_total_mean": 0.21266832947731018, "reward_meter_mean": 0.21266832947731018, "reward_meter_std": 0.20291058719158173, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.21266832947731018, "reward_total_composite_std": 0.20291058719158173} {"timestamp_utc": "2026-04-11T23:14:53Z", "mode": "train", "global_step": 940, "epoch": 0.03775555287785677, "loss": 0.023, "grad_norm": 4.8500847816467285, "learning_rate": 7.154545454545455e-06, "num_tokens": 2104942.0, "completions/mean_length": 115.875, "completions/min_length": 111.0, "completions/max_length": 123.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.875, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.968498945236206, "rewards/meter/std": 0.029964042827486992, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7747991681098938, "rewards/total_composite/std": 0.02397123910486698, "reward": 0.7747991681098938, "reward_std": 0.023971248418092728, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0124904103577137, "sampling/sampling_logp_difference/max": 1.0763840675354004, "sampling/importance_sampling_ratio/min": 0.3408257067203522, "sampling/importance_sampling_ratio/mean": 0.9974149465560913, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.044212615583091974, "clip_ratio/low_mean": 0.0020325202494859695, "clip_ratio/low_min": 0.0020325202494859695, "clip_ratio/high_mean": 0.016339467372745275, "clip_ratio/high_max": 0.016339467372745275, "clip_ratio/region_mean": 0.018371987622231245, "reward_total_mean": 0.7747991681098938, "reward_meter_mean": 0.968498945236206, "reward_meter_std": 0.029964042827486992, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7747991681098938, "reward_total_composite_std": 0.02397123910486698} {"timestamp_utc": "2026-04-11T23:14:57Z", "mode": "train", "global_step": 941, "epoch": 0.037795718359641725, "loss": 0.0096, "grad_norm": 3.194761276245117, "learning_rate": 7.151515151515152e-06, "num_tokens": 2106780.0, "completions/mean_length": 75.75, "completions/min_length": 72.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.75, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.7696533203125, "rewards/meter/std": 0.264241486787796, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.7389180660247803, "rewards/total_composite/std": 0.2821867763996124, "reward": 0.7389180660247803, "reward_std": 0.2821867763996124, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02568567357957363, "sampling/sampling_logp_difference/max": 1.0589550733566284, "sampling/importance_sampling_ratio/min": 0.3468180000782013, "sampling/importance_sampling_ratio/mean": 1.0013823509216309, "sampling/importance_sampling_ratio/max": 1.8680615425109863, "entropy": 0.08911147527396679, "clip_ratio/low_mean": 0.011406926438212395, "clip_ratio/low_min": 0.011406926438212395, "clip_ratio/high_mean": 0.008275165455415845, "clip_ratio/high_max": 0.008275165455415845, "clip_ratio/region_mean": 0.01968209189362824, "reward_total_mean": 0.7389180660247803, "reward_meter_mean": 0.7696533203125, "reward_meter_std": 0.264241486787796, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.7389180660247803, "reward_total_composite_std": 0.2821867763996124} {"timestamp_utc": "2026-04-11T23:15:02Z", "mode": "train", "global_step": 942, "epoch": 0.03783588384142668, "loss": -0.0086, "grad_norm": 11.929455757141113, "learning_rate": 7.148484848484849e-06, "num_tokens": 2108257.0, "completions/mean_length": 38.625, "completions/min_length": 37.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.625, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.7800159454345703, "rewards/meter/std": 0.17752352356910706, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7800159454345703, "rewards/total_composite/std": 0.17752352356910706, "reward": 0.7800159454345703, "reward_std": 0.17752352356910706, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03380492329597473, "sampling/sampling_logp_difference/max": 1.4989476203918457, "sampling/importance_sampling_ratio/min": 0.5866755843162537, "sampling/importance_sampling_ratio/mean": 1.0109392404556274, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15884944330900908, "clip_ratio/low_mean": 0.006667852168902755, "clip_ratio/low_min": 0.006667852168902755, "clip_ratio/high_mean": 0.016215793788433075, "clip_ratio/high_max": 0.016215793788433075, "clip_ratio/region_mean": 0.02288364595733583, "reward_total_mean": 0.7800159454345703, "reward_meter_mean": 0.7800159454345703, "reward_meter_std": 0.17752352356910706, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7800159454345703, "reward_total_composite_std": 0.17752352356910706} {"timestamp_utc": "2026-04-11T23:15:06Z", "mode": "train", "global_step": 943, "epoch": 0.03787604932321163, "loss": -0.0148, "grad_norm": 6.928620338439941, "learning_rate": 7.145454545454547e-06, "num_tokens": 2109890.0, "completions/mean_length": 41.125, "completions/min_length": 39.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.125, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9855058193206787, "rewards/meter/std": 0.010752059519290924, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9855058193206787, "rewards/total_composite/std": 0.010752059519290924, "reward": 0.9855058193206787, "reward_std": 0.010752064175903797, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01714111864566803, "sampling/sampling_logp_difference/max": 1.1181397438049316, "sampling/importance_sampling_ratio/min": 0.326887309551239, "sampling/importance_sampling_ratio/mean": 0.9965353012084961, "sampling/importance_sampling_ratio/max": 1.2362208366394043, "entropy": 0.09372959332540631, "clip_ratio/low_mean": 0.006253908621147275, "clip_ratio/low_min": 0.006253908621147275, "clip_ratio/high_mean": 0.015178571688011289, "clip_ratio/high_max": 0.015178571688011289, "clip_ratio/region_mean": 0.021432480309158564, "reward_total_mean": 0.9855058193206787, "reward_meter_mean": 0.9855058193206787, "reward_meter_std": 0.010752059519290924, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9855058193206787, "reward_total_composite_std": 0.010752059519290924} {"timestamp_utc": "2026-04-11T23:15:12Z", "mode": "train", "global_step": 944, "epoch": 0.03791621480499659, "loss": 0.0057, "grad_norm": 4.29167366027832, "learning_rate": 7.142424242424243e-06, "num_tokens": 2112370.0, "completions/mean_length": 123.0, "completions/min_length": 117.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 123.0, "completions/min_terminated_length": 117.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.8062102794647217, "rewards/meter/std": 0.18132120370864868, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.5850111842155457, "rewards/total_composite/std": 0.11425027996301651, "reward": 0.5850111842155457, "reward_std": 0.11425027996301651, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026136090978980064, "sampling/sampling_logp_difference/max": 1.7183735370635986, "sampling/importance_sampling_ratio/min": 0.1793576329946518, "sampling/importance_sampling_ratio/mean": 0.994314432144165, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06403588084504008, "clip_ratio/low_mean": 0.004050420364364982, "clip_ratio/low_min": 0.004050420364364982, "clip_ratio/high_mean": 0.01517849147785455, "clip_ratio/high_max": 0.01517849147785455, "clip_ratio/region_mean": 0.01922891184221953, "reward_total_mean": 0.5850111842155457, "reward_meter_mean": 0.8062102794647217, "reward_meter_std": 0.18132120370864868, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.5850111842155457, "reward_total_composite_std": 0.11425027996301651} {"timestamp_utc": "2026-04-11T23:15:18Z", "mode": "train", "global_step": 945, "epoch": 0.03795638028678154, "loss": 0.0177, "grad_norm": 4.3193440437316895, "learning_rate": 7.1393939393939405e-06, "num_tokens": 2114655.0, "completions/mean_length": 120.625, "completions/min_length": 119.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.625, "completions/min_terminated_length": 119.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.9766536355018616, "rewards/meter/std": 0.021832915022969246, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7142857313156128, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6976097822189331, "rewards/total_composite/std": 0.015594935044646263, "reward": 0.6976097822189331, "reward_std": 0.015594931319355965, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015354371629655361, "sampling/sampling_logp_difference/max": 1.3104934692382812, "sampling/importance_sampling_ratio/min": 0.2696869373321533, "sampling/importance_sampling_ratio/mean": 0.9987999200820923, "sampling/importance_sampling_ratio/max": 1.6762388944625854, "entropy": 0.05396859138272703, "clip_ratio/low_mean": 0.002040850231423974, "clip_ratio/low_min": 0.002040850231423974, "clip_ratio/high_mean": 0.011334090260788798, "clip_ratio/high_max": 0.011334090260788798, "clip_ratio/region_mean": 0.013374940492212772, "reward_total_mean": 0.6976097822189331, "reward_meter_mean": 0.9766536355018616, "reward_meter_std": 0.021832915022969246, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7142857313156128, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6976097822189331, "reward_total_composite_std": 0.015594935044646263} {"timestamp_utc": "2026-04-11T23:15:24Z", "mode": "train", "global_step": 946, "epoch": 0.037996545768566495, "loss": 0.0042, "grad_norm": 1.7848657369613647, "learning_rate": 7.136363636363637e-06, "num_tokens": 2117343.0, "completions/mean_length": 166.0, "completions/min_length": 159.0, "completions/max_length": 171.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 166.0, "completions/min_terminated_length": 159.0, "completions/max_terminated_length": 171.0, "rewards/meter/mean": 0.9958733320236206, "rewards/meter/std": 0.00040963603532873094, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6428571939468384, "rewards/repeat_penalty/std": 0.1322600245475769, "rewards/total_composite/mean": 0.6401993036270142, "rewards/total_composite/std": 0.13169293105602264, "reward": 0.6401993036270142, "reward_std": 0.13169293105602264, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009576267562806606, "sampling/sampling_logp_difference/max": 1.0388219356536865, "sampling/importance_sampling_ratio/min": 0.35387131571769714, "sampling/importance_sampling_ratio/mean": 0.9997925162315369, "sampling/importance_sampling_ratio/max": 1.6956576108932495, "entropy": 0.03645319910719991, "clip_ratio/low_mean": 0.00222885946277529, "clip_ratio/low_min": 0.00222885946277529, "clip_ratio/high_mean": 0.008308789343573153, "clip_ratio/high_max": 0.008308789343573153, "clip_ratio/region_mean": 0.010537648806348443, "reward_total_mean": 0.6401993036270142, "reward_meter_mean": 0.9958733320236206, "reward_meter_std": 0.00040963603532873094, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6428571939468384, "reward_repeat_penalty_std": 0.1322600245475769, "reward_total_composite_mean": 0.6401993036270142, "reward_total_composite_std": 0.13169293105602264} {"timestamp_utc": "2026-04-11T23:15:31Z", "mode": "train", "global_step": 947, "epoch": 0.03803671125035145, "loss": -0.004, "grad_norm": 1.0996932983398438, "learning_rate": 7.133333333333334e-06, "num_tokens": 2121363.0, "completions/mean_length": 293.5, "completions/min_length": 287.0, "completions/max_length": 306.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 293.5, "completions/min_terminated_length": 287.0, "completions/max_terminated_length": 306.0, "rewards/meter/mean": 0.9976366758346558, "rewards/meter/std": 0.0012180638732388616, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.06681530922651291, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.39835166931152344, "rewards/repeat_penalty/std": 0.22086545825004578, "rewards/total_composite/mean": 0.37909868359565735, "rewards/total_composite/std": 0.22086751461029053, "reward": 0.37909868359565735, "reward_std": 0.22086749970912933, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0038092564791440964, "sampling/sampling_logp_difference/max": 0.7576923370361328, "sampling/importance_sampling_ratio/min": 0.4687469005584717, "sampling/importance_sampling_ratio/mean": 1.0007612705230713, "sampling/importance_sampling_ratio/max": 1.6547387838363647, "entropy": 0.021893693367019296, "clip_ratio/low_mean": 0.0021205995872151107, "clip_ratio/low_min": 0.0021205995872151107, "clip_ratio/high_mean": 0.0017141569405794144, "clip_ratio/high_max": 0.0017141569405794144, "clip_ratio/region_mean": 0.003834756527794525, "reward_total_mean": 0.37909868359565735, "reward_meter_mean": 0.9976366758346558, "reward_meter_std": 0.0012180638732388616, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.06681530922651291, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.39835166931152344, "reward_repeat_penalty_std": 0.22086545825004578, "reward_total_composite_mean": 0.37909868359565735, "reward_total_composite_std": 0.22086751461029053} {"timestamp_utc": "2026-04-11T23:15:35Z", "mode": "train", "global_step": 948, "epoch": 0.0380768767321364, "loss": -0.012, "grad_norm": 6.672044277191162, "learning_rate": 7.130303030303031e-06, "num_tokens": 2123173.0, "completions/mean_length": 70.25, "completions/min_length": 67.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.25, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9970372915267944, "rewards/meter/std": 0.0017233057878911495, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9970372915267944, "rewards/total_composite/std": 0.0017233057878911495, "reward": 0.9970372915267944, "reward_std": 0.001723314169794321, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02216244488954544, "sampling/sampling_logp_difference/max": 0.823767364025116, "sampling/importance_sampling_ratio/min": 0.43877550959587097, "sampling/importance_sampling_ratio/mean": 1.0034829378128052, "sampling/importance_sampling_ratio/max": 1.8399385213851929, "entropy": 0.1040426753461361, "clip_ratio/low_mean": 0.0072228144854307175, "clip_ratio/low_min": 0.0072228144854307175, "clip_ratio/high_mean": 0.008802816737443209, "clip_ratio/high_max": 0.008802816737443209, "clip_ratio/region_mean": 0.016025631222873926, "reward_total_mean": 0.9970372915267944, "reward_meter_mean": 0.9970372915267944, "reward_meter_std": 0.0017233057878911495, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9970372915267944, "reward_total_composite_std": 0.0017233057878911495} {"timestamp_utc": "2026-04-11T23:15:40Z", "mode": "train", "global_step": 949, "epoch": 0.03811704221392136, "loss": 0.0108, "grad_norm": 1.9708746671676636, "learning_rate": 7.127272727272728e-06, "num_tokens": 2125026.0, "completions/mean_length": 60.625, "completions/min_length": 60.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.625, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.7747896909713745, "rewards/meter/std": 0.38887178897857666, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.75, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.5285885334014893, "rewards/total_composite/std": 0.23691311478614807, "reward": 0.5285885334014893, "reward_std": 0.23691309988498688, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0069820573553442955, "sampling/sampling_logp_difference/max": 0.7136227488517761, "sampling/importance_sampling_ratio/min": 0.48986631631851196, "sampling/importance_sampling_ratio/mean": 1.0037261247634888, "sampling/importance_sampling_ratio/max": 1.2806702852249146, "entropy": 0.03217482636682689, "clip_ratio/low_mean": 0.006048386916518211, "clip_ratio/low_min": 0.006048386916518211, "clip_ratio/high_mean": 0.004098360426723957, "clip_ratio/high_max": 0.004098360426723957, "clip_ratio/region_mean": 0.010146747343242168, "reward_total_mean": 0.5285885334014893, "reward_meter_mean": 0.7747896909713745, "reward_meter_std": 0.38887178897857666, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.75, "reward_repeat_penalty_std": 0.15430334210395813, "reward_total_composite_mean": 0.5285885334014893, "reward_total_composite_std": 0.23691311478614807} {"timestamp_utc": "2026-04-11T23:15:45Z", "mode": "train", "global_step": 950, "epoch": 0.03815720769570631, "loss": 0.0008, "grad_norm": 3.3005993366241455, "learning_rate": 7.124242424242424e-06, "num_tokens": 2127299.0, "completions/mean_length": 120.125, "completions/min_length": 117.0, "completions/max_length": 123.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.125, "completions/min_terminated_length": 117.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.9637161493301392, "rewards/meter/std": 0.08138205856084824, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7749999761581421, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.7464566826820374, "rewards/total_composite/std": 0.09103824943304062, "reward": 0.7464566826820374, "reward_std": 0.09103825688362122, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010762141086161137, "sampling/sampling_logp_difference/max": 1.548635721206665, "sampling/importance_sampling_ratio/min": 0.2125377207994461, "sampling/importance_sampling_ratio/mean": 0.996985137462616, "sampling/importance_sampling_ratio/max": 1.577369213104248, "entropy": 0.030512073542922735, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.007258509169332683, "clip_ratio/high_max": 0.007258509169332683, "clip_ratio/region_mean": 0.007258509169332683, "reward_total_mean": 0.7464566826820374, "reward_meter_mean": 0.9637161493301392, "reward_meter_std": 0.08138205856084824, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7749999761581421, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.7464566826820374, "reward_total_composite_std": 0.09103824943304062} {"timestamp_utc": "2026-04-11T23:16:56Z", "mode": "eval", "global_step": 950, "epoch": 0.03815720769570631, "eval_loss": NaN, "eval_runtime": 71.2662, "eval_samples_per_second": 1.459, "eval_steps_per_second": 0.182, "eval_num_tokens": 2127299.0, "eval_completions/mean_length": 205.85576923076923, "eval_completions/min_length": 61.53846153846154, "eval_completions/max_length": 376.0769230769231, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 205.85576923076923, "eval_completions/min_terminated_length": 61.53846153846154, "eval_completions/max_terminated_length": 376.0769230769231, "eval_rewards/meter/mean": 0.6758117584081796, "eval_rewards/meter/std": 0.3830513243491833, "eval_rewards/count_adherence/mean": 0.9363291034331689, "eval_rewards/count_adherence/std": 0.09608005073208076, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/repeat_penalty/mean": 0.6863612211667575, "eval_rewards/repeat_penalty/std": 0.2006635952454347, "eval_rewards/total_composite/mean": 0.41787450359417844, "eval_rewards/total_composite/std": 0.3079184889793396, "eval_reward": 0.41787450359417844, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.0055299800498267776, "eval_sampling/sampling_logp_difference/max": 0.7134297077472394, "eval_sampling/importance_sampling_ratio/min": 0.5113821442310627, "eval_sampling/importance_sampling_ratio/mean": 1.0011087380922759, "eval_sampling/importance_sampling_ratio/max": 1.2753138267076933, "eval_entropy": 0.04716356213276203, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.41787450359417844, "eval_reward_meter_mean": 0.6758117584081796, "eval_reward_meter_std": 0.3830513243491833, "eval_reward_count_adherence_mean": 0.9363291034331689, "eval_reward_count_adherence_std": 0.09608005073208076, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_repeat_penalty_mean": 0.6863612211667575, "eval_reward_repeat_penalty_std": 0.2006635952454347, "eval_reward_total_composite_mean": 0.41787450359417844, "eval_reward_total_composite_std": 0.3079184889793396} {"timestamp_utc": "2026-04-11T23:17:03Z", "mode": "train", "global_step": 951, "epoch": 0.038197373177491264, "loss": 0.0095, "grad_norm": 4.512762069702148, "learning_rate": 7.121212121212122e-06, "num_tokens": 2129201.0, "completions/mean_length": 66.75, "completions/min_length": 63.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.75, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.2129497528076172, "rewards/meter/std": 0.24503695964813232, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.2129497528076172, "rewards/total_composite/std": 0.24503695964813232, "reward": 0.2129497528076172, "reward_std": 0.24503694474697113, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02744590863585472, "sampling/sampling_logp_difference/max": 1.3395442962646484, "sampling/importance_sampling_ratio/min": 0.2619650065898895, "sampling/importance_sampling_ratio/mean": 1.001453161239624, "sampling/importance_sampling_ratio/max": 1.650530457496643, "entropy": 0.21755263209342957, "clip_ratio/low_mean": 0.00940205657389015, "clip_ratio/low_min": 0.00940205657389015, "clip_ratio/high_mean": 0.011278833262622356, "clip_ratio/high_max": 0.011278833262622356, "clip_ratio/region_mean": 0.020680889836512506, "reward_total_mean": 0.2129497528076172, "reward_meter_mean": 0.2129497528076172, "reward_meter_std": 0.24503695964813232, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.2129497528076172, "reward_total_composite_std": 0.24503695964813232} {"timestamp_utc": "2026-04-11T23:17:13Z", "mode": "train", "global_step": 952, "epoch": 0.03823753865927622, "loss": -0.0019, "grad_norm": 0.9411554336547852, "learning_rate": 7.118181818181819e-06, "num_tokens": 2134537.0, "completions/mean_length": 451.0, "completions/min_length": 446.0, "completions/max_length": 470.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 451.0, "completions/min_terminated_length": 446.0, "completions/max_terminated_length": 470.0, "rewards/meter/mean": 0.49129414558410645, "rewards/meter/std": 0.5206316113471985, "rewards/count_adherence/mean": 0.6315789222717285, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5652173757553101, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.17538189888000488, "rewards/total_composite/std": 0.18585476279258728, "reward": 0.17538189888000488, "reward_std": 0.18585477769374847, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004929275717586279, "sampling/sampling_logp_difference/max": 3.009263277053833, "sampling/importance_sampling_ratio/min": 0.04932800680398941, "sampling/importance_sampling_ratio/mean": 1.0005483627319336, "sampling/importance_sampling_ratio/max": 1.997017741203308, "entropy": 0.013566843350417912, "clip_ratio/low_mean": 0.0016632305341772735, "clip_ratio/low_min": 0.0016632305341772735, "clip_ratio/high_mean": 0.0016673028003424406, "clip_ratio/high_max": 0.0016673028003424406, "clip_ratio/region_mean": 0.003330533334519714, "reward_total_mean": 0.17538189888000488, "reward_meter_mean": 0.49129414558410645, "reward_meter_std": 0.5206316113471985, "reward_count_adherence_mean": 0.6315789222717285, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5652173757553101, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.17538189888000488, "reward_total_composite_std": 0.18585476279258728} {"timestamp_utc": "2026-04-11T23:17:18Z", "mode": "train", "global_step": 953, "epoch": 0.03827770414106117, "loss": 0.0166, "grad_norm": 7.106107711791992, "learning_rate": 7.115151515151516e-06, "num_tokens": 2136613.0, "completions/mean_length": 80.5, "completions/min_length": 78.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.5, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9900895357131958, "rewards/meter/std": 0.006374072283506393, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.8412514925003052, "rewards/total_composite/std": 0.08799233287572861, "reward": 0.8412514925003052, "reward_std": 0.08799233287572861, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03556607663631439, "sampling/sampling_logp_difference/max": 2.8349223136901855, "sampling/importance_sampling_ratio/min": 0.0587230883538723, "sampling/importance_sampling_ratio/mean": 1.0007519721984863, "sampling/importance_sampling_ratio/max": 1.8997235298156738, "entropy": 0.09662660211324692, "clip_ratio/low_mean": 0.02908552496228367, "clip_ratio/low_min": 0.02908552496228367, "clip_ratio/high_mean": 0.00769954826682806, "clip_ratio/high_max": 0.00769954826682806, "clip_ratio/region_mean": 0.03678507322911173, "reward_total_mean": 0.8412514925003052, "reward_meter_mean": 0.9900895357131958, "reward_meter_std": 0.006374072283506393, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.8412514925003052, "reward_total_composite_std": 0.08799233287572861} {"timestamp_utc": "2026-04-11T23:17:23Z", "mode": "train", "global_step": 954, "epoch": 0.038317869622846126, "loss": 0.0104, "grad_norm": 3.721968650817871, "learning_rate": 7.1121212121212125e-06, "num_tokens": 2138876.0, "completions/mean_length": 104.875, "completions/min_length": 99.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 104.875, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.28197556734085083, "rewards/meter/std": 0.30284935235977173, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7749999761581421, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.22144606709480286, "rewards/total_composite/std": 0.24437181651592255, "reward": 0.22144606709480286, "reward_std": 0.24437181651592255, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.029477477073669434, "sampling/sampling_logp_difference/max": 1.4413394927978516, "sampling/importance_sampling_ratio/min": 0.23661062121391296, "sampling/importance_sampling_ratio/mean": 1.00920569896698, "sampling/importance_sampling_ratio/max": 1.7154247760772705, "entropy": 0.17342047207057476, "clip_ratio/low_mean": 0.0235287812538445, "clip_ratio/low_min": 0.0235287812538445, "clip_ratio/high_mean": 0.009797494392842054, "clip_ratio/high_max": 0.009797494392842054, "clip_ratio/region_mean": 0.033326275646686554, "reward_total_mean": 0.22144606709480286, "reward_meter_mean": 0.28197556734085083, "reward_meter_std": 0.30284935235977173, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7749999761581421, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.22144606709480286, "reward_total_composite_std": 0.24437181651592255} {"timestamp_utc": "2026-04-11T23:17:29Z", "mode": "train", "global_step": 955, "epoch": 0.03835803510463108, "loss": 0.009, "grad_norm": 0.7482571005821228, "learning_rate": 7.10909090909091e-06, "num_tokens": 2141066.0, "completions/mean_length": 113.75, "completions/min_length": 113.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 113.75, "completions/min_terminated_length": 113.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.969186544418335, "rewards/meter/std": 0.005740889813750982, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7753492593765259, "rewards/total_composite/std": 0.0045927222818136215, "reward": 0.7753492593765259, "reward_std": 0.0045927101746201515, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003761922474950552, "sampling/sampling_logp_difference/max": 0.406435489654541, "sampling/importance_sampling_ratio/min": 0.729997992515564, "sampling/importance_sampling_ratio/mean": 1.0022902488708496, "sampling/importance_sampling_ratio/max": 1.501456379890442, "entropy": 0.02124447817914188, "clip_ratio/low_mean": 0.005445147980935872, "clip_ratio/low_min": 0.005445147980935872, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.005445147980935872, "reward_total_mean": 0.7753492593765259, "reward_meter_mean": 0.969186544418335, "reward_meter_std": 0.005740889813750982, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7753492593765259, "reward_total_composite_std": 0.0045927222818136215} {"timestamp_utc": "2026-04-11T23:17:34Z", "mode": "train", "global_step": 956, "epoch": 0.038398200586416034, "loss": 0.014, "grad_norm": 3.8508245944976807, "learning_rate": 7.106060606060606e-06, "num_tokens": 2143130.0, "completions/mean_length": 104.0, "completions/min_length": 101.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 104.0, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.9977437853813171, "rewards/meter/std": 0.0006820790586061776, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.8979662656784058, "rewards/total_composite/std": 0.10663474351167679, "reward": 0.8979662656784058, "reward_std": 0.10663473606109619, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026750473305583, "sampling/sampling_logp_difference/max": 2.033867359161377, "sampling/importance_sampling_ratio/min": 0.1308285892009735, "sampling/importance_sampling_ratio/mean": 1.0018119812011719, "sampling/importance_sampling_ratio/max": 1.998322606086731, "entropy": 0.07065040059387684, "clip_ratio/low_mean": 0.0059523810632526875, "clip_ratio/low_min": 0.0059523810632526875, "clip_ratio/high_mean": 0.0036183277843520045, "clip_ratio/high_max": 0.0036183277843520045, "clip_ratio/region_mean": 0.009570708847604692, "reward_total_mean": 0.8979662656784058, "reward_meter_mean": 0.9977437853813171, "reward_meter_std": 0.0006820790586061776, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_total_composite_mean": 0.8979662656784058, "reward_total_composite_std": 0.10663474351167679} {"timestamp_utc": "2026-04-11T23:17:38Z", "mode": "train", "global_step": 957, "epoch": 0.03843836606820099, "loss": 0.033, "grad_norm": 6.492129802703857, "learning_rate": 7.103030303030304e-06, "num_tokens": 2145169.0, "completions/mean_length": 60.875, "completions/min_length": 60.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.875, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9347407817840576, "rewards/meter/std": 0.14110930263996124, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7083333730697632, "rewards/repeat_penalty/std": 0.11785111576318741, "rewards/total_composite/mean": 0.6475568413734436, "rewards/total_composite/std": 0.025069493800401688, "reward": 0.6475568413734436, "reward_std": 0.02506948821246624, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003160211257636547, "sampling/sampling_logp_difference/max": 0.3890615701675415, "sampling/importance_sampling_ratio/min": 0.6776925325393677, "sampling/importance_sampling_ratio/mean": 0.9995921850204468, "sampling/importance_sampling_ratio/max": 1.1547554731369019, "entropy": 0.01487105863634497, "clip_ratio/low_mean": 0.0037313431967049837, "clip_ratio/low_min": 0.0037313431967049837, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0037313431967049837, "reward_total_mean": 0.6475568413734436, "reward_meter_mean": 0.9347407817840576, "reward_meter_std": 0.14110930263996124, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7083333730697632, "reward_repeat_penalty_std": 0.11785111576318741, "reward_total_composite_mean": 0.6475568413734436, "reward_total_composite_std": 0.025069493800401688} {"timestamp_utc": "2026-04-11T23:17:43Z", "mode": "train", "global_step": 958, "epoch": 0.03847853154998594, "loss": -0.0039, "grad_norm": 10.342998504638672, "learning_rate": 7.100000000000001e-06, "num_tokens": 2146649.0, "completions/mean_length": 35.0, "completions/min_length": 34.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.4885302782058716, "rewards/meter/std": 0.35837680101394653, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4885302782058716, "rewards/total_composite/std": 0.35837680101394653, "reward": 0.4885302782058716, "reward_std": 0.35837680101394653, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03277422487735748, "sampling/sampling_logp_difference/max": 1.3705358505249023, "sampling/importance_sampling_ratio/min": 0.25397083163261414, "sampling/importance_sampling_ratio/mean": 0.9996507167816162, "sampling/importance_sampling_ratio/max": 1.507129430770874, "entropy": 0.10543786082416773, "clip_ratio/low_mean": 0.010714286006987095, "clip_ratio/low_min": 0.010714286006987095, "clip_ratio/high_mean": 0.010615079430863261, "clip_ratio/high_max": 0.010615079430863261, "clip_ratio/region_mean": 0.021329365437850356, "reward_total_mean": 0.4885302782058716, "reward_meter_mean": 0.4885302782058716, "reward_meter_std": 0.35837680101394653, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.4885302782058716, "reward_total_composite_std": 0.35837680101394653} {"timestamp_utc": "2026-04-11T23:17:50Z", "mode": "train", "global_step": 959, "epoch": 0.038518697031770896, "loss": -0.0351, "grad_norm": 1.4002562761306763, "learning_rate": 7.096969696969698e-06, "num_tokens": 2150789.0, "completions/mean_length": 294.5, "completions/min_length": 274.0, "completions/max_length": 310.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 294.5, "completions/min_terminated_length": 274.0, "completions/max_terminated_length": 310.0, "rewards/meter/mean": 0.9977169036865234, "rewards/meter/std": 0.0006972014671191573, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4833333492279053, "rewards/repeat_penalty/std": 0.216024711728096, "rewards/total_composite/mean": 0.4822729825973511, "rewards/total_composite/std": 0.21558886766433716, "reward": 0.4822729825973511, "reward_std": 0.21558886766433716, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0083305099979043, "sampling/sampling_logp_difference/max": 1.0299842357635498, "sampling/importance_sampling_ratio/min": 0.3570125997066498, "sampling/importance_sampling_ratio/mean": 0.9994634389877319, "sampling/importance_sampling_ratio/max": 1.689242959022522, "entropy": 0.04552340344525874, "clip_ratio/low_mean": 0.001365295291179791, "clip_ratio/low_min": 0.001365295291179791, "clip_ratio/high_mean": 0.006645367917371914, "clip_ratio/high_max": 0.006645367917371914, "clip_ratio/region_mean": 0.008010663208551705, "reward_total_mean": 0.4822729825973511, "reward_meter_mean": 0.9977169036865234, "reward_meter_std": 0.0006972014671191573, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4833333492279053, "reward_repeat_penalty_std": 0.216024711728096, "reward_total_composite_mean": 0.4822729825973511, "reward_total_composite_std": 0.21558886766433716} {"timestamp_utc": "2026-04-11T23:17:56Z", "mode": "train", "global_step": 960, "epoch": 0.03855886251355585, "loss": -0.0083, "grad_norm": 6.993642807006836, "learning_rate": 7.093939393939394e-06, "num_tokens": 2153531.0, "completions/mean_length": 160.75, "completions/min_length": 156.0, "completions/max_length": 165.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 160.75, "completions/min_terminated_length": 156.0, "completions/max_terminated_length": 165.0, "rewards/meter/mean": 0.36679452657699585, "rewards/meter/std": 0.16856136918067932, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7222222089767456, "rewards/repeat_penalty/std": 0.08399210125207901, "rewards/total_composite/mean": 0.2650865912437439, "rewards/total_composite/std": 0.1350317746400833, "reward": 0.2650865912437439, "reward_std": 0.1350317746400833, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02235065959393978, "sampling/sampling_logp_difference/max": 4.669411659240723, "sampling/importance_sampling_ratio/min": 0.00937778502702713, "sampling/importance_sampling_ratio/mean": 0.9979313015937805, "sampling/importance_sampling_ratio/max": 1.8480039834976196, "entropy": 0.05065230838954449, "clip_ratio/low_mean": 0.009524673456326127, "clip_ratio/low_min": 0.009524673456326127, "clip_ratio/high_mean": 0.009170350269414485, "clip_ratio/high_max": 0.009170350269414485, "clip_ratio/region_mean": 0.01869502372574061, "reward_total_mean": 0.2650865912437439, "reward_meter_mean": 0.36679452657699585, "reward_meter_std": 0.16856136918067932, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7222222089767456, "reward_repeat_penalty_std": 0.08399210125207901, "reward_total_composite_mean": 0.2650865912437439, "reward_total_composite_std": 0.1350317746400833} {"timestamp_utc": "2026-04-11T23:18:01Z", "mode": "train", "global_step": 961, "epoch": 0.038599027995340804, "loss": 0.0198, "grad_norm": 5.430922508239746, "learning_rate": 7.0909090909090916e-06, "num_tokens": 2155316.0, "completions/mean_length": 65.125, "completions/min_length": 62.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.125, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9629546403884888, "rewards/meter/std": 0.05961780995130539, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9629546403884888, "rewards/total_composite/std": 0.05961780995130539, "reward": 0.9629546403884888, "reward_std": 0.05961783230304718, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024385379627346992, "sampling/sampling_logp_difference/max": 1.6775139570236206, "sampling/importance_sampling_ratio/min": 0.18683788180351257, "sampling/importance_sampling_ratio/mean": 0.9952221512794495, "sampling/importance_sampling_ratio/max": 1.2267403602600098, "entropy": 0.0778669067658484, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.01166338287293911, "clip_ratio/high_max": 0.01166338287293911, "clip_ratio/region_mean": 0.01166338287293911, "reward_total_mean": 0.9629546403884888, "reward_meter_mean": 0.9629546403884888, "reward_meter_std": 0.05961780995130539, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9629546403884888, "reward_total_composite_std": 0.05961780995130539} {"timestamp_utc": "2026-04-11T23:18:05Z", "mode": "train", "global_step": 962, "epoch": 0.03863919347712576, "loss": 0.0499, "grad_norm": 8.99122142791748, "learning_rate": 7.087878787878788e-06, "num_tokens": 2157171.0, "completions/mean_length": 65.875, "completions/min_length": 62.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.875, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.32342612743377686, "rewards/meter/std": 0.291739284992218, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.32342612743377686, "rewards/total_composite/std": 0.291739284992218, "reward": 0.32342612743377686, "reward_std": 0.291739284992218, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04449179768562317, "sampling/sampling_logp_difference/max": 1.642319679260254, "sampling/importance_sampling_ratio/min": 0.1935305893421173, "sampling/importance_sampling_ratio/mean": 1.0009416341781616, "sampling/importance_sampling_ratio/max": 1.8445723056793213, "entropy": 0.24257362261414528, "clip_ratio/low_mean": 0.01819333794992417, "clip_ratio/low_min": 0.01819333794992417, "clip_ratio/high_mean": 0.015785872004926205, "clip_ratio/high_max": 0.015785872004926205, "clip_ratio/region_mean": 0.033979209954850376, "reward_total_mean": 0.32342612743377686, "reward_meter_mean": 0.32342612743377686, "reward_meter_std": 0.291739284992218, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.32342612743377686, "reward_total_composite_std": 0.291739284992218} {"timestamp_utc": "2026-04-11T23:18:15Z", "mode": "train", "global_step": 963, "epoch": 0.03867935895891071, "loss": 0.0162, "grad_norm": 1.0686217546463013, "learning_rate": 7.084848484848485e-06, "num_tokens": 2162491.0, "completions/mean_length": 445.0, "completions/min_length": 435.0, "completions/max_length": 468.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 445.0, "completions/min_terminated_length": 435.0, "completions/max_terminated_length": 468.0, "rewards/meter/mean": 0.8052209615707397, "rewards/meter/std": 0.37245726585388184, "rewards/count_adherence/mean": 0.7291666269302368, "rewards/count_adherence/std": 0.01964186504483223, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6346794962882996, "rewards/repeat_penalty/std": 0.046481478959321976, "rewards/total_composite/mean": 0.3684366047382355, "rewards/total_composite/std": 0.16659744083881378, "reward": 0.3684366047382355, "reward_std": 0.1665974259376526, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010675698518753052, "sampling/sampling_logp_difference/max": 2.1844980716705322, "sampling/importance_sampling_ratio/min": 0.11253420263528824, "sampling/importance_sampling_ratio/mean": 0.9988970160484314, "sampling/importance_sampling_ratio/max": 1.6414300203323364, "entropy": 0.03673372324556112, "clip_ratio/low_mean": 0.002477022586390376, "clip_ratio/low_min": 0.002477022586390376, "clip_ratio/high_mean": 0.0073644123040139675, "clip_ratio/high_max": 0.0073644123040139675, "clip_ratio/region_mean": 0.009841434890404344, "reward_total_mean": 0.3684366047382355, "reward_meter_mean": 0.8052209615707397, "reward_meter_std": 0.37245726585388184, "reward_count_adherence_mean": 0.7291666269302368, "reward_count_adherence_std": 0.01964186504483223, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6346794962882996, "reward_repeat_penalty_std": 0.046481478959321976, "reward_total_composite_mean": 0.3684366047382355, "reward_total_composite_std": 0.16659744083881378} {"timestamp_utc": "2026-04-11T23:18:22Z", "mode": "train", "global_step": 964, "epoch": 0.038719524440695666, "loss": -0.005, "grad_norm": 2.381939649581909, "learning_rate": 7.081818181818182e-06, "num_tokens": 2165977.0, "completions/mean_length": 240.75, "completions/min_length": 234.0, "completions/max_length": 261.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 240.75, "completions/min_terminated_length": 234.0, "completions/max_terminated_length": 261.0, "rewards/meter/mean": 0.9859161376953125, "rewards/meter/std": 0.0027882950380444527, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6057692766189575, "rewards/repeat_penalty/std": 0.027196412906050682, "rewards/total_composite/mean": 0.5972245931625366, "rewards/total_composite/std": 0.026564769446849823, "reward": 0.5972245931625366, "reward_std": 0.026564771309494972, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004197314847260714, "sampling/sampling_logp_difference/max": 1.4283943176269531, "sampling/importance_sampling_ratio/min": 0.23969349265098572, "sampling/importance_sampling_ratio/mean": 1.0001258850097656, "sampling/importance_sampling_ratio/max": 1.6210657358169556, "entropy": 0.012361971399514005, "clip_ratio/low_mean": 0.0005319148767739534, "clip_ratio/low_min": 0.0005319148767739534, "clip_ratio/high_mean": 0.0026147099561057985, "clip_ratio/high_max": 0.0026147099561057985, "clip_ratio/region_mean": 0.003146624832879752, "reward_total_mean": 0.5972245931625366, "reward_meter_mean": 0.9859161376953125, "reward_meter_std": 0.0027882950380444527, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6057692766189575, "reward_repeat_penalty_std": 0.027196412906050682, "reward_total_composite_mean": 0.5972245931625366, "reward_total_composite_std": 0.026564769446849823} {"timestamp_utc": "2026-04-11T23:18:27Z", "mode": "train", "global_step": 965, "epoch": 0.03875968992248062, "loss": 0.0083, "grad_norm": 6.1013593673706055, "learning_rate": 7.07878787878788e-06, "num_tokens": 2168080.0, "completions/mean_length": 71.875, "completions/min_length": 68.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.875, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9859936833381653, "rewards/meter/std": 0.00813794881105423, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9859936833381653, "rewards/total_composite/std": 0.00813794881105423, "reward": 0.9859936833381653, "reward_std": 0.008137939497828484, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008848754689097404, "sampling/sampling_logp_difference/max": 0.8346943855285645, "sampling/importance_sampling_ratio/min": 0.6168099045753479, "sampling/importance_sampling_ratio/mean": 1.001380443572998, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03642029734328389, "clip_ratio/low_mean": 0.0036498295376077294, "clip_ratio/low_min": 0.0036498295376077294, "clip_ratio/high_mean": 0.005208333372138441, "clip_ratio/high_max": 0.005208333372138441, "clip_ratio/region_mean": 0.00885816290974617, "reward_total_mean": 0.9859936833381653, "reward_meter_mean": 0.9859936833381653, "reward_meter_std": 0.00813794881105423, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9859936833381653, "reward_total_composite_std": 0.00813794881105423} {"timestamp_utc": "2026-04-11T23:18:31Z", "mode": "train", "global_step": 966, "epoch": 0.038799855404265574, "loss": 0.0052, "grad_norm": 10.40634822845459, "learning_rate": 7.075757575757576e-06, "num_tokens": 2169886.0, "completions/mean_length": 64.75, "completions/min_length": 62.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.75, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.7862817645072937, "rewards/meter/std": 0.3110516667366028, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7862817645072937, "rewards/total_composite/std": 0.3110516667366028, "reward": 0.7862817645072937, "reward_std": 0.3110516369342804, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05011482164263725, "sampling/sampling_logp_difference/max": 1.9868323802947998, "sampling/importance_sampling_ratio/min": 0.3482150137424469, "sampling/importance_sampling_ratio/mean": 1.0035208463668823, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.18688618578016758, "clip_ratio/low_mean": 0.005769230891019106, "clip_ratio/low_min": 0.005769230891019106, "clip_ratio/high_mean": 0.048127192771062255, "clip_ratio/high_max": 0.048127192771062255, "clip_ratio/region_mean": 0.05389642366208136, "reward_total_mean": 0.7862817645072937, "reward_meter_mean": 0.7862817645072937, "reward_meter_std": 0.3110516667366028, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7862817645072937, "reward_total_composite_std": 0.3110516667366028} {"timestamp_utc": "2026-04-11T23:18:38Z", "mode": "train", "global_step": 967, "epoch": 0.03884002088605053, "loss": -0.0072, "grad_norm": 2.4970195293426514, "learning_rate": 7.072727272727273e-06, "num_tokens": 2172905.0, "completions/mean_length": 194.375, "completions/min_length": 184.0, "completions/max_length": 206.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 194.375, "completions/min_terminated_length": 184.0, "completions/max_terminated_length": 206.0, "rewards/meter/mean": 0.5914201736450195, "rewards/meter/std": 0.4835823178291321, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6805555820465088, "rewards/repeat_penalty/std": 0.03928370773792267, "rewards/total_composite/mean": 0.39438170194625854, "rewards/total_composite/std": 0.32224810123443604, "reward": 0.39438170194625854, "reward_std": 0.32224807143211365, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008972376585006714, "sampling/sampling_logp_difference/max": 1.8997198343276978, "sampling/importance_sampling_ratio/min": 0.1496105194091797, "sampling/importance_sampling_ratio/mean": 1.000235676765442, "sampling/importance_sampling_ratio/max": 1.642437219619751, "entropy": 0.03619017521850765, "clip_ratio/low_mean": 0.005190499185118824, "clip_ratio/low_min": 0.005190499185118824, "clip_ratio/high_mean": 0.004360174119938165, "clip_ratio/high_max": 0.004360174119938165, "clip_ratio/region_mean": 0.00955067330505699, "reward_total_mean": 0.39438170194625854, "reward_meter_mean": 0.5914201736450195, "reward_meter_std": 0.4835823178291321, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6805555820465088, "reward_repeat_penalty_std": 0.03928370773792267, "reward_total_composite_mean": 0.39438170194625854, "reward_total_composite_std": 0.32224810123443604} {"timestamp_utc": "2026-04-11T23:18:43Z", "mode": "train", "global_step": 968, "epoch": 0.03888018636783548, "loss": -0.0085, "grad_norm": 5.905569076538086, "learning_rate": 7.06969696969697e-06, "num_tokens": 2174629.0, "completions/mean_length": 60.5, "completions/min_length": 58.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.5, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.1817864626646042, "rewards/meter/std": 0.293150931596756, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.1817864626646042, "rewards/total_composite/std": 0.293150931596756, "reward": 0.1817864626646042, "reward_std": 0.293150931596756, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026289377361536026, "sampling/sampling_logp_difference/max": 0.7234911918640137, "sampling/importance_sampling_ratio/min": 0.4889802932739258, "sampling/importance_sampling_ratio/mean": 1.0088768005371094, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0875935684889555, "clip_ratio/low_mean": 0.03520799265243113, "clip_ratio/low_min": 0.03520799265243113, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.03520799265243113, "reward_total_mean": 0.1817864626646042, "reward_meter_mean": 0.1817864626646042, "reward_meter_std": 0.293150931596756, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.1817864626646042, "reward_total_composite_std": 0.293150931596756} {"timestamp_utc": "2026-04-11T23:18:48Z", "mode": "train", "global_step": 969, "epoch": 0.038920351849620435, "loss": -0.0153, "grad_norm": 3.71621036529541, "learning_rate": 7.066666666666667e-06, "num_tokens": 2176683.0, "completions/mean_length": 93.75, "completions/min_length": 87.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.75, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.5728106498718262, "rewards/meter/std": 0.2251601219177246, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.824999988079071, "rewards/repeat_penalty/std": 0.12817399203777313, "rewards/total_composite/mean": 0.4796490967273712, "rewards/total_composite/std": 0.21056687831878662, "reward": 0.4796490967273712, "reward_std": 0.21056686341762543, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024638189002871513, "sampling/sampling_logp_difference/max": 1.6777713298797607, "sampling/importance_sampling_ratio/min": 0.18678979575634003, "sampling/importance_sampling_ratio/mean": 0.9970004558563232, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07108194520696998, "clip_ratio/low_mean": 0.009395235683768988, "clip_ratio/low_min": 0.009395235683768988, "clip_ratio/high_mean": 0.01694070769008249, "clip_ratio/high_max": 0.01694070769008249, "clip_ratio/region_mean": 0.026335943373851478, "reward_total_mean": 0.4796490967273712, "reward_meter_mean": 0.5728106498718262, "reward_meter_std": 0.2251601219177246, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.824999988079071, "reward_repeat_penalty_std": 0.12817399203777313, "reward_total_composite_mean": 0.4796490967273712, "reward_total_composite_std": 0.21056687831878662} {"timestamp_utc": "2026-04-11T23:18:53Z", "mode": "train", "global_step": 970, "epoch": 0.03896051733140539, "loss": -0.0005, "grad_norm": 2.2865495681762695, "learning_rate": 7.063636363636365e-06, "num_tokens": 2178721.0, "completions/mean_length": 78.75, "completions/min_length": 77.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.75, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9977834224700928, "rewards/meter/std": 0.00039716452010907233, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977834224700928, "rewards/total_composite/std": 0.00039716452010907233, "reward": 0.9977834224700928, "reward_std": 0.0003971673722844571, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019730689004063606, "sampling/sampling_logp_difference/max": 0.7925920486450195, "sampling/importance_sampling_ratio/min": 0.4526699483394623, "sampling/importance_sampling_ratio/mean": 1.0018593072891235, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09748060069978237, "clip_ratio/low_mean": 0.011199243483133614, "clip_ratio/low_min": 0.011199243483133614, "clip_ratio/high_mean": 0.006389971007592976, "clip_ratio/high_max": 0.006389971007592976, "clip_ratio/region_mean": 0.01758921449072659, "reward_total_mean": 0.9977834224700928, "reward_meter_mean": 0.9977834224700928, "reward_meter_std": 0.00039716452010907233, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9977834224700928, "reward_total_composite_std": 0.00039716452010907233} {"timestamp_utc": "2026-04-11T23:18:59Z", "mode": "train", "global_step": 971, "epoch": 0.03900068281319034, "loss": -0.0062, "grad_norm": 1.4529743194580078, "learning_rate": 7.060606060606061e-06, "num_tokens": 2181793.0, "completions/mean_length": 189.0, "completions/min_length": 183.0, "completions/max_length": 192.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 189.0, "completions/min_terminated_length": 183.0, "completions/max_terminated_length": 192.0, "rewards/meter/mean": 0.9985735416412354, "rewards/meter/std": 0.0002435932110529393, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6527777910232544, "rewards/repeat_penalty/std": 0.13849149644374847, "rewards/total_composite/mean": 0.6518542766571045, "rewards/total_composite/std": 0.13831683993339539, "reward": 0.6518542766571045, "reward_std": 0.13831683993339539, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016793714836239815, "sampling/sampling_logp_difference/max": 5.936721324920654, "sampling/importance_sampling_ratio/min": 0.0026406734250485897, "sampling/importance_sampling_ratio/mean": 0.998883068561554, "sampling/importance_sampling_ratio/max": 1.8489798307418823, "entropy": 0.06829563481733203, "clip_ratio/low_mean": 0.002027027076110244, "clip_ratio/low_min": 0.002027027076110244, "clip_ratio/high_mean": 0.010603155242279172, "clip_ratio/high_max": 0.010603155242279172, "clip_ratio/region_mean": 0.012630182318389416, "reward_total_mean": 0.6518542766571045, "reward_meter_mean": 0.9985735416412354, "reward_meter_std": 0.0002435932110529393, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6527777910232544, "reward_repeat_penalty_std": 0.13849149644374847, "reward_total_composite_mean": 0.6518542766571045, "reward_total_composite_std": 0.13831683993339539} {"timestamp_utc": "2026-04-11T23:19:06Z", "mode": "train", "global_step": 972, "epoch": 0.0390408482949753, "loss": -0.01, "grad_norm": 1.3656582832336426, "learning_rate": 7.057575757575759e-06, "num_tokens": 2185373.0, "completions/mean_length": 263.5, "completions/min_length": 248.0, "completions/max_length": 273.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 263.5, "completions/min_terminated_length": 248.0, "completions/max_terminated_length": 273.0, "rewards/meter/mean": 0.6416140794754028, "rewards/meter/std": 0.4567407965660095, "rewards/count_adherence/mean": 0.859375, "rewards/count_adherence/std": 0.04419417306780815, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4212246239185333, "rewards/repeat_penalty/std": 0.21046686172485352, "rewards/total_composite/mean": 0.19842997193336487, "rewards/total_composite/std": 0.19151148200035095, "reward": 0.19842997193336487, "reward_std": 0.19151146709918976, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017121732234954834, "sampling/sampling_logp_difference/max": 7.143291473388672, "sampling/importance_sampling_ratio/min": 0.0007901470526121557, "sampling/importance_sampling_ratio/mean": 1.0005074739456177, "sampling/importance_sampling_ratio/max": 1.6406102180480957, "entropy": 0.05923895724117756, "clip_ratio/low_mean": 0.006205761310411617, "clip_ratio/low_min": 0.006205761310411617, "clip_ratio/high_mean": 0.0037591225700452924, "clip_ratio/high_max": 0.0037591225700452924, "clip_ratio/region_mean": 0.00996488388045691, "reward_total_mean": 0.19842997193336487, "reward_meter_mean": 0.6416140794754028, "reward_meter_std": 0.4567407965660095, "reward_count_adherence_mean": 0.859375, "reward_count_adherence_std": 0.04419417306780815, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4212246239185333, "reward_repeat_penalty_std": 0.21046686172485352, "reward_total_composite_mean": 0.19842997193336487, "reward_total_composite_std": 0.19151148200035095} {"timestamp_utc": "2026-04-11T23:19:11Z", "mode": "train", "global_step": 973, "epoch": 0.03908101377676025, "loss": 0.0171, "grad_norm": 13.225342750549316, "learning_rate": 7.054545454545455e-06, "num_tokens": 2187067.0, "completions/mean_length": 60.75, "completions/min_length": 60.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.75, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.5853145122528076, "rewards/meter/std": 0.1321393847465515, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5853145122528076, "rewards/total_composite/std": 0.1321393847465515, "reward": 0.5853145122528076, "reward_std": 0.13213936984539032, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022791719064116478, "sampling/sampling_logp_difference/max": 1.2052631378173828, "sampling/importance_sampling_ratio/min": 0.2996131479740143, "sampling/importance_sampling_ratio/mean": 0.9999557733535767, "sampling/importance_sampling_ratio/max": 1.8551063537597656, "entropy": 0.09659606032073498, "clip_ratio/low_mean": 0.008165820967406034, "clip_ratio/low_min": 0.008165820967406034, "clip_ratio/high_mean": 0.010416667209938169, "clip_ratio/high_max": 0.010416667209938169, "clip_ratio/region_mean": 0.018582488177344203, "reward_total_mean": 0.5853145122528076, "reward_meter_mean": 0.5853145122528076, "reward_meter_std": 0.1321393847465515, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5853145122528076, "reward_total_composite_std": 0.1321393847465515} {"timestamp_utc": "2026-04-11T23:19:15Z", "mode": "train", "global_step": 974, "epoch": 0.039121179258545205, "loss": 0.0027, "grad_norm": 2.125204563140869, "learning_rate": 7.0515151515151525e-06, "num_tokens": 2188986.0, "completions/mean_length": 78.875, "completions/min_length": 78.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.875, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9978776574134827, "rewards/meter/std": 0.0004215049266349524, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978776574134827, "rewards/total_composite/std": 0.0004215049266349524, "reward": 0.9978776574134827, "reward_std": 0.0004215273365844041, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013205158524215221, "sampling/sampling_logp_difference/max": 0.6857354640960693, "sampling/importance_sampling_ratio/min": 0.503719687461853, "sampling/importance_sampling_ratio/mean": 1.0038025379180908, "sampling/importance_sampling_ratio/max": 1.3348405361175537, "entropy": 0.08664691727608442, "clip_ratio/low_mean": 0.006329114083200693, "clip_ratio/low_min": 0.006329114083200693, "clip_ratio/high_mean": 0.004807692370377481, "clip_ratio/high_max": 0.004807692370377481, "clip_ratio/region_mean": 0.011136806453578174, "reward_total_mean": 0.9978776574134827, "reward_meter_mean": 0.9978776574134827, "reward_meter_std": 0.0004215049266349524, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9978776574134827, "reward_total_composite_std": 0.0004215049266349524} {"timestamp_utc": "2026-04-11T23:19:20Z", "mode": "train", "global_step": 975, "epoch": 0.03916134474033016, "loss": 0.0108, "grad_norm": 2.8000571727752686, "learning_rate": 7.048484848484849e-06, "num_tokens": 2191193.0, "completions/mean_length": 100.875, "completions/min_length": 97.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.875, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.996698796749115, "rewards/meter/std": 0.0009329493041150272, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9249999523162842, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.921911358833313, "rewards/total_composite/std": 0.10283299535512924, "reward": 0.921911358833313, "reward_std": 0.10283298790454865, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027565594762563705, "sampling/sampling_logp_difference/max": 1.987597942352295, "sampling/importance_sampling_ratio/min": 0.1370241641998291, "sampling/importance_sampling_ratio/mean": 0.9971323609352112, "sampling/importance_sampling_ratio/max": 1.8002057075500488, "entropy": 0.08865811070427299, "clip_ratio/low_mean": 0.003676470718346536, "clip_ratio/low_min": 0.003676470718346536, "clip_ratio/high_mean": 0.019024619832634926, "clip_ratio/high_max": 0.019024619832634926, "clip_ratio/region_mean": 0.022701090550981462, "reward_total_mean": 0.921911358833313, "reward_meter_mean": 0.996698796749115, "reward_meter_std": 0.0009329493041150272, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9249999523162842, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.921911358833313, "reward_total_composite_std": 0.10283299535512924} {"timestamp_utc": "2026-04-11T23:19:25Z", "mode": "train", "global_step": 976, "epoch": 0.03920151022211511, "loss": -0.0053, "grad_norm": 8.02316665649414, "learning_rate": 7.045454545454546e-06, "num_tokens": 2193051.0, "completions/mean_length": 60.25, "completions/min_length": 60.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.25, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9850078821182251, "rewards/meter/std": 0.0014023992698639631, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6566718816757202, "rewards/total_composite/std": 0.0009349193423986435, "reward": 0.6566718816757202, "reward_std": 0.0009349363390356302, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009783357381820679, "sampling/sampling_logp_difference/max": 0.7054405212402344, "sampling/importance_sampling_ratio/min": 0.4981825351715088, "sampling/importance_sampling_ratio/mean": 1.0041868686676025, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.029942457331344485, "clip_ratio/low_mean": 0.006250000325962901, "clip_ratio/low_min": 0.006250000325962901, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.006250000325962901, "reward_total_mean": 0.6566718816757202, "reward_meter_mean": 0.9850078821182251, "reward_meter_std": 0.0014023992698639631, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6566718816757202, "reward_total_composite_std": 0.0009349193423986435} {"timestamp_utc": "2026-04-11T23:19:30Z", "mode": "train", "global_step": 977, "epoch": 0.03924167570390007, "loss": -0.0065, "grad_norm": 1.635718584060669, "learning_rate": 7.0424242424242426e-06, "num_tokens": 2194984.0, "completions/mean_length": 79.625, "completions/min_length": 79.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.625, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9982393980026245, "rewards/meter/std": 0.0004721701261587441, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9982393980026245, "rewards/total_composite/std": 0.0004721701261587441, "reward": 0.9982393980026245, "reward_std": 0.00047218104009516537, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013445981778204441, "sampling/sampling_logp_difference/max": 0.5942535400390625, "sampling/importance_sampling_ratio/min": 0.5519744157791138, "sampling/importance_sampling_ratio/mean": 1.0005873441696167, "sampling/importance_sampling_ratio/max": 1.298281192779541, "entropy": 0.09024440124630928, "clip_ratio/low_mean": 0.006289557088166475, "clip_ratio/low_min": 0.006289557088166475, "clip_ratio/high_mean": 0.007735339575447142, "clip_ratio/high_max": 0.007735339575447142, "clip_ratio/region_mean": 0.014024896663613617, "reward_total_mean": 0.9982393980026245, "reward_meter_mean": 0.9982393980026245, "reward_meter_std": 0.0004721701261587441, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9982393980026245, "reward_total_composite_std": 0.0004721701261587441} {"timestamp_utc": "2026-04-11T23:19:35Z", "mode": "train", "global_step": 978, "epoch": 0.03928184118568502, "loss": 0.0049, "grad_norm": 3.430257558822632, "learning_rate": 7.039393939393941e-06, "num_tokens": 2196933.0, "completions/mean_length": 84.625, "completions/min_length": 82.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.625, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.99366694688797, "rewards/meter/std": 0.0010994295589625835, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7949336171150208, "rewards/total_composite/std": 0.0008795479079708457, "reward": 0.7949336171150208, "reward_std": 0.0008795478497631848, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016848498955368996, "sampling/sampling_logp_difference/max": 1.4610930681228638, "sampling/importance_sampling_ratio/min": 0.23198257386684418, "sampling/importance_sampling_ratio/mean": 0.9972406029701233, "sampling/importance_sampling_ratio/max": 1.2917805910110474, "entropy": 0.06632238999009132, "clip_ratio/low_mean": 0.0029248768696561456, "clip_ratio/low_min": 0.0029248768696561456, "clip_ratio/high_mean": 0.008984935469925404, "clip_ratio/high_max": 0.008984935469925404, "clip_ratio/region_mean": 0.01190981233958155, "reward_total_mean": 0.7949336171150208, "reward_meter_mean": 0.99366694688797, "reward_meter_std": 0.0010994295589625835, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7949336171150208, "reward_total_composite_std": 0.0008795479079708457} {"timestamp_utc": "2026-04-11T23:19:39Z", "mode": "train", "global_step": 979, "epoch": 0.039322006667469975, "loss": -0.0028, "grad_norm": 6.046172618865967, "learning_rate": 7.036363636363637e-06, "num_tokens": 2198744.0, "completions/mean_length": 60.375, "completions/min_length": 59.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.375, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.6340106725692749, "rewards/meter/std": 0.20943842828273773, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6340106725692749, "rewards/total_composite/std": 0.20943842828273773, "reward": 0.6340106725692749, "reward_std": 0.20943844318389893, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030584923923015594, "sampling/sampling_logp_difference/max": 1.2826333045959473, "sampling/importance_sampling_ratio/min": 0.27730610966682434, "sampling/importance_sampling_ratio/mean": 0.9989505410194397, "sampling/importance_sampling_ratio/max": 1.586508870124817, "entropy": 0.1247731251642108, "clip_ratio/low_mean": 0.016670140204951167, "clip_ratio/low_min": 0.016670140204951167, "clip_ratio/high_mean": 0.020312500651925802, "clip_ratio/high_max": 0.020312500651925802, "clip_ratio/region_mean": 0.03698264085687697, "reward_total_mean": 0.6340106725692749, "reward_meter_mean": 0.6340106725692749, "reward_meter_std": 0.20943842828273773, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6340106725692749, "reward_total_composite_std": 0.20943842828273773} {"timestamp_utc": "2026-04-11T23:19:45Z", "mode": "train", "global_step": 980, "epoch": 0.03936217214925493, "loss": -0.0001, "grad_norm": 1.3390476703643799, "learning_rate": 7.033333333333334e-06, "num_tokens": 2201037.0, "completions/mean_length": 117.625, "completions/min_length": 116.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 117.625, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.9986953139305115, "rewards/meter/std": 0.00018651132995728403, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7989562153816223, "rewards/total_composite/std": 0.00014920807734597474, "reward": 0.7989562153816223, "reward_std": 0.00014920474495738745, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012999390251934528, "sampling/sampling_logp_difference/max": 1.3501521348953247, "sampling/importance_sampling_ratio/min": 0.2592008113861084, "sampling/importance_sampling_ratio/mean": 1.001077651977539, "sampling/importance_sampling_ratio/max": 1.392466425895691, "entropy": 0.0817764438688755, "clip_ratio/low_mean": 0.005235925782471895, "clip_ratio/low_min": 0.005235925782471895, "clip_ratio/high_mean": 0.0010683761211112142, "clip_ratio/high_max": 0.0010683761211112142, "clip_ratio/region_mean": 0.006304301903583109, "reward_total_mean": 0.7989562153816223, "reward_meter_mean": 0.9986953139305115, "reward_meter_std": 0.00018651132995728403, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7989562153816223, "reward_total_composite_std": 0.00014920807734597474} {"timestamp_utc": "2026-04-11T23:19:49Z", "mode": "train", "global_step": 981, "epoch": 0.03940233763103988, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.030303030303031e-06, "num_tokens": 2202789.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9872130751609802, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6581420302391052, "rewards/total_composite/std": 0.0, "reward": 0.6581420302391052, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0014747538371011615, "sampling/sampling_logp_difference/max": 0.05223120003938675, "sampling/importance_sampling_ratio/min": 0.9491093754768372, "sampling/importance_sampling_ratio/mean": 1.001099944114685, "sampling/importance_sampling_ratio/max": 1.0427228212356567, "entropy": 0.012774199014529586, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.6581420302391052, "reward_meter_mean": 0.9872130751609802, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6581420302391052, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-11T23:19:54Z", "mode": "train", "global_step": 982, "epoch": 0.03944250311282484, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.027272727272728e-06, "num_tokens": 2204653.0, "completions/mean_length": 52.0, "completions/min_length": 52.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.0, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.961655855178833, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.961655855178833, "rewards/total_composite/std": 0.0, "reward": 0.961655855178833, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.007805653847754002, "sampling/sampling_logp_difference/max": 0.8052540421485901, "sampling/importance_sampling_ratio/min": 0.44697439670562744, "sampling/importance_sampling_ratio/mean": 1.0008008480072021, "sampling/importance_sampling_ratio/max": 1.1538965702056885, "entropy": 0.04045943822711706, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.961655855178833, "reward_meter_mean": 0.961655855178833, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.961655855178833, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-11T23:19:59Z", "mode": "train", "global_step": 983, "epoch": 0.03948266859460979, "loss": -0.0039, "grad_norm": 2.655090093612671, "learning_rate": 7.024242424242424e-06, "num_tokens": 2206569.0, "completions/mean_length": 80.5, "completions/min_length": 79.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.5, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.998711347579956, "rewards/meter/std": 0.0004273413505870849, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998711347579956, "rewards/total_composite/std": 0.0004273413505870849, "reward": 0.998711347579956, "reward_std": 0.000427335879066959, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01396770216524601, "sampling/sampling_logp_difference/max": 0.902503490447998, "sampling/importance_sampling_ratio/min": 0.40555310249328613, "sampling/importance_sampling_ratio/mean": 1.0012972354888916, "sampling/importance_sampling_ratio/max": 1.6649136543273926, "entropy": 0.07731548231095076, "clip_ratio/low_mean": 0.0015625000232830644, "clip_ratio/low_min": 0.0015625000232830644, "clip_ratio/high_mean": 0.0030487803742289543, "clip_ratio/high_max": 0.0030487803742289543, "clip_ratio/region_mean": 0.004611280397512019, "reward_total_mean": 0.998711347579956, "reward_meter_mean": 0.998711347579956, "reward_meter_std": 0.0004273413505870849, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.998711347579956, "reward_total_composite_std": 0.0004273413505870849} {"timestamp_utc": "2026-04-11T23:20:03Z", "mode": "train", "global_step": 984, "epoch": 0.039522834076394744, "loss": -0.0007, "grad_norm": 4.302158832550049, "learning_rate": 7.021212121212122e-06, "num_tokens": 2208124.0, "completions/mean_length": 31.375, "completions/min_length": 31.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.375, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9924632906913757, "rewards/meter/std": 0.0021250685676932335, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9924632906913757, "rewards/total_composite/std": 0.0021250685676932335, "reward": 0.9924632906913757, "reward_std": 0.0021250564604997635, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014777499251067638, "sampling/sampling_logp_difference/max": 0.5788311958312988, "sampling/importance_sampling_ratio/min": 0.7517499923706055, "sampling/importance_sampling_ratio/mean": 1.0098501443862915, "sampling/importance_sampling_ratio/max": 1.7839521169662476, "entropy": 0.08878939691931009, "clip_ratio/low_mean": 0.016003023833036423, "clip_ratio/low_min": 0.016003023833036423, "clip_ratio/high_mean": 0.007938507944345474, "clip_ratio/high_max": 0.007938507944345474, "clip_ratio/region_mean": 0.023941531777381897, "reward_total_mean": 0.9924632906913757, "reward_meter_mean": 0.9924632906913757, "reward_meter_std": 0.0021250685676932335, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9924632906913757, "reward_total_composite_std": 0.0021250685676932335} {"timestamp_utc": "2026-04-11T23:20:08Z", "mode": "train", "global_step": 985, "epoch": 0.0395629995581797, "loss": -0.0119, "grad_norm": 8.724225997924805, "learning_rate": 7.018181818181818e-06, "num_tokens": 2209934.0, "completions/mean_length": 61.25, "completions/min_length": 61.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.25, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9870957136154175, "rewards/meter/std": 0.00033180107129737735, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7083333730697632, "rewards/repeat_penalty/std": 0.11785111576318741, "rewards/total_composite/mean": 0.6991585493087769, "rewards/total_composite/std": 0.11601238697767258, "reward": 0.6991585493087769, "reward_std": 0.11601236462593079, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012023198418319225, "sampling/sampling_logp_difference/max": 1.0977474451065063, "sampling/importance_sampling_ratio/min": 0.3336217403411865, "sampling/importance_sampling_ratio/mean": 1.0018260478973389, "sampling/importance_sampling_ratio/max": 1.6728845834732056, "entropy": 0.04339231993071735, "clip_ratio/low_mean": 0.020491802133619785, "clip_ratio/low_min": 0.020491802133619785, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.020491802133619785, "reward_total_mean": 0.6991585493087769, "reward_meter_mean": 0.9870957136154175, "reward_meter_std": 0.00033180107129737735, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7083333730697632, "reward_repeat_penalty_std": 0.11785111576318741, "reward_total_composite_mean": 0.6991585493087769, "reward_total_composite_std": 0.11601238697767258} {"timestamp_utc": "2026-04-11T23:20:13Z", "mode": "train", "global_step": 986, "epoch": 0.03960316503996465, "loss": -0.0012, "grad_norm": 10.352254867553711, "learning_rate": 7.015151515151516e-06, "num_tokens": 2212176.0, "completions/mean_length": 99.25, "completions/min_length": 94.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.25, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.9944287538528442, "rewards/meter/std": 0.0026359334588050842, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9695063829421997, "rewards/total_composite/std": 0.06953759491443634, "reward": 0.9695063829421997, "reward_std": 0.06953759491443634, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.044116295874118805, "sampling/sampling_logp_difference/max": 2.196290969848633, "sampling/importance_sampling_ratio/min": 0.11121489107608795, "sampling/importance_sampling_ratio/mean": 1.007062315940857, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17434827517718077, "clip_ratio/low_mean": 0.003989361692219973, "clip_ratio/low_min": 0.003989361692219973, "clip_ratio/high_mean": 0.037494323682039976, "clip_ratio/high_max": 0.037494323682039976, "clip_ratio/region_mean": 0.04148368537425995, "reward_total_mean": 0.9695063829421997, "reward_meter_mean": 0.9944287538528442, "reward_meter_std": 0.0026359334588050842, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9695063829421997, "reward_total_composite_std": 0.06953759491443634} {"timestamp_utc": "2026-04-11T23:20:18Z", "mode": "train", "global_step": 987, "epoch": 0.039643330521749606, "loss": -0.0152, "grad_norm": 4.8408732414245605, "learning_rate": 7.0121212121212126e-06, "num_tokens": 2213841.0, "completions/mean_length": 61.125, "completions/min_length": 58.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.125, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9843782782554626, "rewards/meter/std": 0.00654714647680521, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7916666865348816, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.7787973880767822, "rewards/total_composite/std": 0.16700218617916107, "reward": 0.7787973880767822, "reward_std": 0.16700218617916107, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02119199000298977, "sampling/sampling_logp_difference/max": 1.5137114524841309, "sampling/importance_sampling_ratio/min": 0.22009159624576569, "sampling/importance_sampling_ratio/mean": 0.9993690252304077, "sampling/importance_sampling_ratio/max": 1.5299021005630493, "entropy": 0.08241967670619488, "clip_ratio/low_mean": 0.022379032103344798, "clip_ratio/low_min": 0.022379032103344798, "clip_ratio/high_mean": 0.0040023052133619785, "clip_ratio/high_max": 0.0040023052133619785, "clip_ratio/region_mean": 0.026381337316706777, "reward_total_mean": 0.7787973880767822, "reward_meter_mean": 0.9843782782554626, "reward_meter_std": 0.00654714647680521, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7916666865348816, "reward_repeat_penalty_std": 0.17251639068126678, "reward_total_composite_mean": 0.7787973880767822, "reward_total_composite_std": 0.16700218617916107} {"timestamp_utc": "2026-04-11T23:20:23Z", "mode": "train", "global_step": 988, "epoch": 0.03968349600353456, "loss": 0.0136, "grad_norm": 3.5085859298706055, "learning_rate": 7.00909090909091e-06, "num_tokens": 2215857.0, "completions/mean_length": 89.0, "completions/min_length": 87.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.0, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9934116005897522, "rewards/meter/std": 0.0031344336457550526, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.8443044424057007, "rewards/total_composite/std": 0.09088363498449326, "reward": 0.8443044424057007, "reward_std": 0.09088362008333206, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023283695802092552, "sampling/sampling_logp_difference/max": 1.422020673751831, "sampling/importance_sampling_ratio/min": 0.24122609198093414, "sampling/importance_sampling_ratio/mean": 1.0044476985931396, "sampling/importance_sampling_ratio/max": 1.6077567338943481, "entropy": 0.10643400624394417, "clip_ratio/low_mean": 0.019569998257793486, "clip_ratio/low_min": 0.019569998257793486, "clip_ratio/high_mean": 0.007151253987103701, "clip_ratio/high_max": 0.007151253987103701, "clip_ratio/region_mean": 0.026721252244897187, "reward_total_mean": 0.8443044424057007, "reward_meter_mean": 0.9934116005897522, "reward_meter_std": 0.0031344336457550526, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.8443044424057007, "reward_total_composite_std": 0.09088363498449326} {"timestamp_utc": "2026-04-11T23:20:28Z", "mode": "train", "global_step": 989, "epoch": 0.039723661485319514, "loss": 0.0136, "grad_norm": 4.891083717346191, "learning_rate": 7.006060606060606e-06, "num_tokens": 2217566.0, "completions/mean_length": 61.625, "completions/min_length": 61.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.625, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9817750453948975, "rewards/meter/std": 0.007981428876519203, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9817750453948975, "rewards/total_composite/std": 0.007981428876519203, "reward": 0.9817750453948975, "reward_std": 0.00798142608255148, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02616567350924015, "sampling/sampling_logp_difference/max": 1.2843780517578125, "sampling/importance_sampling_ratio/min": 0.27682268619537354, "sampling/importance_sampling_ratio/mean": 1.0027326345443726, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.14261625241488218, "clip_ratio/low_mean": 0.01203381922096014, "clip_ratio/low_min": 0.01203381922096014, "clip_ratio/high_mean": 0.014212056528776884, "clip_ratio/high_max": 0.014212056528776884, "clip_ratio/region_mean": 0.026245875749737024, "reward_total_mean": 0.9817750453948975, "reward_meter_mean": 0.9817750453948975, "reward_meter_std": 0.007981428876519203, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9817750453948975, "reward_total_composite_std": 0.007981428876519203} {"timestamp_utc": "2026-04-11T23:20:33Z", "mode": "train", "global_step": 990, "epoch": 0.03976382696710447, "loss": 0.0012, "grad_norm": 5.835471153259277, "learning_rate": 7.0030303030303035e-06, "num_tokens": 2219326.0, "completions/mean_length": 60.0, "completions/min_length": 59.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.0, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9907355308532715, "rewards/meter/std": 0.002091656206175685, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9495816826820374, "rewards/total_composite/std": 0.11764256656169891, "reward": 0.9495816826820374, "reward_std": 0.11764255911111832, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024150043725967407, "sampling/sampling_logp_difference/max": 1.4593958854675293, "sampling/importance_sampling_ratio/min": 0.2323766052722931, "sampling/importance_sampling_ratio/mean": 1.0002789497375488, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09698301088064909, "clip_ratio/low_mean": 0.0042372881434857845, "clip_ratio/low_min": 0.0042372881434857845, "clip_ratio/high_mean": 0.022955394815653563, "clip_ratio/high_max": 0.022955394815653563, "clip_ratio/region_mean": 0.027192682959139347, "reward_total_mean": 0.9495816826820374, "reward_meter_mean": 0.9907355308532715, "reward_meter_std": 0.002091656206175685, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.9495816826820374, "reward_total_composite_std": 0.11764256656169891} {"timestamp_utc": "2026-04-11T23:20:42Z", "mode": "train", "global_step": 991, "epoch": 0.03980399244888942, "loss": 0.0039, "grad_norm": 1.7695684432983398, "learning_rate": 7e-06, "num_tokens": 2224258.0, "completions/mean_length": 407.5, "completions/min_length": 394.0, "completions/max_length": 411.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 407.5, "completions/min_terminated_length": 394.0, "completions/max_terminated_length": 411.0, "rewards/meter/mean": 0.9920412302017212, "rewards/meter/std": 0.003196289064362645, "rewards/count_adherence/mean": 0.7272727489471436, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.46064817905426025, "rewards/repeat_penalty/std": 0.15420734882354736, "rewards/total_composite/mean": 0.3326081335544586, "rewards/total_composite/std": 0.11179135739803314, "reward": 0.3326081335544586, "reward_std": 0.11179134249687195, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01644131913781166, "sampling/sampling_logp_difference/max": 3.4832382202148438, "sampling/importance_sampling_ratio/min": 0.030707810074090958, "sampling/importance_sampling_ratio/mean": 0.9987042546272278, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.027644895017147064, "clip_ratio/low_mean": 0.0006097560981288552, "clip_ratio/low_min": 0.0006097560981288552, "clip_ratio/high_mean": 0.008049219497479498, "clip_ratio/high_max": 0.008049219497479498, "clip_ratio/region_mean": 0.008658975595608354, "reward_total_mean": 0.3326081335544586, "reward_meter_mean": 0.9920412302017212, "reward_meter_std": 0.003196289064362645, "reward_count_adherence_mean": 0.7272727489471436, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.46064817905426025, "reward_repeat_penalty_std": 0.15420734882354736, "reward_total_composite_mean": 0.3326081335544586, "reward_total_composite_std": 0.11179135739803314} {"timestamp_utc": "2026-04-11T23:20:51Z", "mode": "train", "global_step": 992, "epoch": 0.039844157930674376, "loss": 0.0087, "grad_norm": 1.6267327070236206, "learning_rate": 6.996969696969698e-06, "num_tokens": 2229368.0, "completions/mean_length": 419.75, "completions/min_length": 409.0, "completions/max_length": 428.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 419.75, "completions/min_terminated_length": 409.0, "completions/max_terminated_length": 428.0, "rewards/meter/mean": 0.9450836181640625, "rewards/meter/std": 0.07978925108909607, "rewards/count_adherence/mean": 0.7954545021057129, "rewards/count_adherence/std": 0.04208271950483322, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.62321937084198, "rewards/repeat_penalty/std": 0.04523347690701485, "rewards/total_composite/mean": 0.4672620892524719, "rewards/total_composite/std": 0.044083379209041595, "reward": 0.4672620892524719, "reward_std": 0.04408339038491249, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01929199881851673, "sampling/sampling_logp_difference/max": 2.7198145389556885, "sampling/importance_sampling_ratio/min": 0.0658869743347168, "sampling/importance_sampling_ratio/mean": 1.000062108039856, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07821797719225287, "clip_ratio/low_mean": 0.0047512390883639455, "clip_ratio/low_min": 0.0047512390883639455, "clip_ratio/high_mean": 0.01013427460566163, "clip_ratio/high_max": 0.01013427460566163, "clip_ratio/region_mean": 0.014885513694025576, "reward_total_mean": 0.4672620892524719, "reward_meter_mean": 0.9450836181640625, "reward_meter_std": 0.07978925108909607, "reward_count_adherence_mean": 0.7954545021057129, "reward_count_adherence_std": 0.04208271950483322, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.62321937084198, "reward_repeat_penalty_std": 0.04523347690701485, "reward_total_composite_mean": 0.4672620892524719, "reward_total_composite_std": 0.044083379209041595} {"timestamp_utc": "2026-04-11T23:20:55Z", "mode": "train", "global_step": 993, "epoch": 0.03988432341245933, "loss": -0.0038, "grad_norm": 5.988306522369385, "learning_rate": 6.993939393939394e-06, "num_tokens": 2231169.0, "completions/mean_length": 63.125, "completions/min_length": 61.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.125, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.6225343942642212, "rewards/meter/std": 0.3685222268104553, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6225343942642212, "rewards/total_composite/std": 0.3685222268104553, "reward": 0.6225343942642212, "reward_std": 0.3685222268104553, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02788170613348484, "sampling/sampling_logp_difference/max": 1.1369147300720215, "sampling/importance_sampling_ratio/min": 0.3208072781562805, "sampling/importance_sampling_ratio/mean": 1.0002118349075317, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1402796907350421, "clip_ratio/low_mean": 0.01408805197570473, "clip_ratio/low_min": 0.01408805197570473, "clip_ratio/high_mean": 0.021799394860863686, "clip_ratio/high_max": 0.021799394860863686, "clip_ratio/region_mean": 0.035887446836568415, "reward_total_mean": 0.6225343942642212, "reward_meter_mean": 0.6225343942642212, "reward_meter_std": 0.3685222268104553, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6225343942642212, "reward_total_composite_std": 0.3685222268104553} {"timestamp_utc": "2026-04-11T23:21:06Z", "mode": "train", "global_step": 994, "epoch": 0.039924488894244284, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.990909090909092e-06, "num_tokens": 2232881.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9936738014221191, "rewards/meter/std": 0.003524529282003641, "rewards/count_adherence/mean": 0.8088235855102539, "rewards/count_adherence/std": 0.027230001986026764, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5536096096038818, "rewards/repeat_penalty/std": 0.012604749761521816, "rewards/total_composite/mean": 0.44484955072402954, "rewards/total_composite/std": 0.015336754731833935, "reward": 0.44484955072402954, "reward_std": 0.015336750075221062, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.44484955072402954, "reward_meter_mean": 0.9936738014221191, "reward_meter_std": 0.003524529282003641, "reward_count_adherence_mean": 0.8088235855102539, "reward_count_adherence_std": 0.027230001986026764, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5536096096038818, "reward_repeat_penalty_std": 0.012604749761521816, "reward_total_composite_mean": 0.44484955072402954, "reward_total_composite_std": 0.015336754731833935} {"timestamp_utc": "2026-04-11T23:21:11Z", "mode": "train", "global_step": 995, "epoch": 0.03996465437602924, "loss": 0.0319, "grad_norm": 5.63449764251709, "learning_rate": 6.987878787878788e-06, "num_tokens": 2234891.0, "completions/mean_length": 76.25, "completions/min_length": 72.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.25, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.03228817135095596, "rewards/meter/std": 0.06514540314674377, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.03228817135095596, "rewards/total_composite/std": 0.06514540314674377, "reward": 0.03228817135095596, "reward_std": 0.06514539569616318, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0636926218867302, "sampling/sampling_logp_difference/max": 2.1695494651794434, "sampling/importance_sampling_ratio/min": 0.11422906816005707, "sampling/importance_sampling_ratio/mean": 1.0099796056747437, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.28803054243326187, "clip_ratio/low_mean": 0.03413326805457473, "clip_ratio/low_min": 0.03413326805457473, "clip_ratio/high_mean": 0.010277777910232544, "clip_ratio/high_max": 0.010277777910232544, "clip_ratio/region_mean": 0.04441104596480727, "reward_total_mean": 0.03228817135095596, "reward_meter_mean": 0.03228817135095596, "reward_meter_std": 0.06514540314674377, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.03228817135095596, "reward_total_composite_std": 0.06514540314674377} {"timestamp_utc": "2026-04-11T23:21:15Z", "mode": "train", "global_step": 996, "epoch": 0.04000481985781419, "loss": 0.0295, "grad_norm": 6.409210205078125, "learning_rate": 6.984848484848485e-06, "num_tokens": 2236814.0, "completions/mean_length": 76.375, "completions/min_length": 73.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.375, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.010214247740805149, "rewards/meter/std": 0.007614050526171923, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.010214247740805149, "rewards/total_composite/std": 0.007614050526171923, "reward": 0.010214247740805149, "reward_std": 0.007614050526171923, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0485016293823719, "sampling/sampling_logp_difference/max": 1.5220353603363037, "sampling/importance_sampling_ratio/min": 0.21826720237731934, "sampling/importance_sampling_ratio/mean": 1.0045915842056274, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23532197065651417, "clip_ratio/low_mean": 0.019090469810180366, "clip_ratio/low_min": 0.019090469810180366, "clip_ratio/high_mean": 0.01975645322818309, "clip_ratio/high_max": 0.01975645322818309, "clip_ratio/region_mean": 0.03884692303836346, "reward_total_mean": 0.010214247740805149, "reward_meter_mean": 0.010214247740805149, "reward_meter_std": 0.007614050526171923, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.010214247740805149, "reward_total_composite_std": 0.007614050526171923} {"timestamp_utc": "2026-04-11T23:21:20Z", "mode": "train", "global_step": 997, "epoch": 0.040044985339599146, "loss": 0.0187, "grad_norm": 5.538641929626465, "learning_rate": 6.981818181818183e-06, "num_tokens": 2238755.0, "completions/mean_length": 77.625, "completions/min_length": 72.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.625, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.12371634691953659, "rewards/meter/std": 0.3405037224292755, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.002925167791545391, "rewards/total_composite/std": 0.005130887497216463, "reward": 0.002925167791545391, "reward_std": 0.005130887031555176, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06538589298725128, "sampling/sampling_logp_difference/max": 4.094723701477051, "sampling/importance_sampling_ratio/min": 0.016660349443554878, "sampling/importance_sampling_ratio/mean": 0.9927118420600891, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15790804475545883, "clip_ratio/low_mean": 0.03696049621794373, "clip_ratio/low_min": 0.03696049621794373, "clip_ratio/high_mean": 0.009801336331292987, "clip_ratio/high_max": 0.009801336331292987, "clip_ratio/region_mean": 0.046761832549236715, "reward_total_mean": 0.002925167791545391, "reward_meter_mean": 0.12371634691953659, "reward_meter_std": 0.3405037224292755, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.002925167791545391, "reward_total_composite_std": 0.005130887497216463} {"timestamp_utc": "2026-04-11T23:21:25Z", "mode": "train", "global_step": 998, "epoch": 0.0400851508213841, "loss": -0.0461, "grad_norm": 7.545363903045654, "learning_rate": 6.978787878787879e-06, "num_tokens": 2240295.0, "completions/mean_length": 42.5, "completions/min_length": 37.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.5, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.9984815120697021, "rewards/meter/std": 0.001035381923429668, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9984815120697021, "rewards/total_composite/std": 0.001035381923429668, "reward": 0.9984815120697021, "reward_std": 0.0010353842517361045, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.043179791420698166, "sampling/sampling_logp_difference/max": 1.6201910972595215, "sampling/importance_sampling_ratio/min": 0.19786088168621063, "sampling/importance_sampling_ratio/mean": 0.9990119934082031, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2211740417405963, "clip_ratio/low_mean": 0.009268068009987473, "clip_ratio/low_min": 0.009268068009987473, "clip_ratio/high_mean": 0.020238095661625266, "clip_ratio/high_max": 0.020238095661625266, "clip_ratio/region_mean": 0.02950616367161274, "reward_total_mean": 0.9984815120697021, "reward_meter_mean": 0.9984815120697021, "reward_meter_std": 0.001035381923429668, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9984815120697021, "reward_total_composite_std": 0.001035381923429668} {"timestamp_utc": "2026-04-11T23:21:30Z", "mode": "train", "global_step": 999, "epoch": 0.040125316303169054, "loss": 0.0238, "grad_norm": 7.812422752380371, "learning_rate": 6.975757575757577e-06, "num_tokens": 2242408.0, "completions/mean_length": 99.125, "completions/min_length": 97.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.125, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.8570353984832764, "rewards/meter/std": 0.18506589531898499, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8570353984832764, "rewards/total_composite/std": 0.18506589531898499, "reward": 0.8570353984832764, "reward_std": 0.18506589531898499, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.039642542600631714, "sampling/sampling_logp_difference/max": 1.2252435684204102, "sampling/importance_sampling_ratio/min": 0.2936861515045166, "sampling/importance_sampling_ratio/mean": 1.0009180307388306, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.22251357324421406, "clip_ratio/low_mean": 0.003737623686902225, "clip_ratio/low_min": 0.003737623686902225, "clip_ratio/high_mean": 0.026683697244152427, "clip_ratio/high_max": 0.026683697244152427, "clip_ratio/region_mean": 0.030421320931054652, "reward_total_mean": 0.8570353984832764, "reward_meter_mean": 0.8570353984832764, "reward_meter_std": 0.18506589531898499, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8570353984832764, "reward_total_composite_std": 0.18506589531898499} {"timestamp_utc": "2026-04-11T23:21:35Z", "mode": "train", "global_step": 1000, "epoch": 0.04016548178495401, "loss": 0.016, "grad_norm": 2.9554319381713867, "learning_rate": 6.9727272727272735e-06, "num_tokens": 2244489.0, "completions/mean_length": 89.125, "completions/min_length": 86.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.125, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.9947108030319214, "rewards/meter/std": 0.004218456335365772, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8250000476837158, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8204019665718079, "rewards/total_composite/std": 0.06665877252817154, "reward": 0.8204019665718079, "reward_std": 0.06665877997875214, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015147675760090351, "sampling/sampling_logp_difference/max": 1.3051207065582275, "sampling/importance_sampling_ratio/min": 0.27113983035087585, "sampling/importance_sampling_ratio/mean": 1.0028539896011353, "sampling/importance_sampling_ratio/max": 1.6871954202651978, "entropy": 0.0689164032228291, "clip_ratio/low_mean": 0.01114646252244711, "clip_ratio/low_min": 0.01114646252244711, "clip_ratio/high_mean": 0.0014534883666783571, "clip_ratio/high_max": 0.0014534883666783571, "clip_ratio/region_mean": 0.012599950889125466, "reward_total_mean": 0.8204019665718079, "reward_meter_mean": 0.9947108030319214, "reward_meter_std": 0.004218456335365772, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8250000476837158, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.8204019665718079, "reward_total_composite_std": 0.06665877252817154} {"timestamp_utc": "2026-04-11T23:22:58Z", "mode": "eval", "global_step": 1000, "epoch": 0.04016548178495401, "eval_loss": NaN, "eval_runtime": 83.6634, "eval_samples_per_second": 1.243, "eval_steps_per_second": 0.155, "eval_num_tokens": 2244489.0, "eval_completions/mean_length": 233.69230769230768, "eval_completions/min_length": 60.0, "eval_completions/max_length": 449.38461538461536, "eval_completions/clipped_ratio": 0.04807692307692308, "eval_completions/mean_terminated_length": 219.62088364821213, "eval_completions/min_terminated_length": 60.0, "eval_completions/max_terminated_length": 413.2307692307692, "eval_rewards/meter/mean": 0.644585329752702, "eval_rewards/meter/std": 0.44141573172349197, "eval_rewards/count_adherence/mean": 0.9180091665341303, "eval_rewards/count_adherence/std": 0.09919005589416394, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/repeat_penalty/mean": 0.7226548378284161, "eval_rewards/repeat_penalty/std": 0.17840037838770792, "eval_rewards/total_composite/mean": 0.42624477927501386, "eval_rewards/total_composite/std": 0.3310052202298091, "eval_reward": 0.42624477927501386, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.00861521104637247, "eval_sampling/sampling_logp_difference/max": 0.8023634048608633, "eval_sampling/importance_sampling_ratio/min": 0.4674220451941857, "eval_sampling/importance_sampling_ratio/mean": 1.0016976503225474, "eval_sampling/importance_sampling_ratio/max": 1.3331910921977117, "eval_entropy": 0.07349483840740643, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.42624477927501386, "eval_reward_meter_mean": 0.644585329752702, "eval_reward_meter_std": 0.44141573172349197, "eval_reward_count_adherence_mean": 0.9180091665341303, "eval_reward_count_adherence_std": 0.09919005589416394, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_repeat_penalty_mean": 0.7226548378284161, "eval_reward_repeat_penalty_std": 0.17840037838770792, "eval_reward_total_composite_mean": 0.42624477927501386, "eval_reward_total_composite_std": 0.3310052202298091} {"timestamp_utc": "2026-04-11T23:23:06Z", "mode": "train", "global_step": 1001, "epoch": 0.04020564726673896, "loss": 0.0233, "grad_norm": 8.061637878417969, "learning_rate": 6.969696969696971e-06, "num_tokens": 2246187.0, "completions/mean_length": 62.25, "completions/min_length": 62.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.25, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.8264527320861816, "rewards/meter/std": 0.1885237991809845, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8264527320861816, "rewards/total_composite/std": 0.1885237991809845, "reward": 0.8264527320861816, "reward_std": 0.1885237991809845, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02593499980866909, "sampling/sampling_logp_difference/max": 1.3253092765808105, "sampling/importance_sampling_ratio/min": 0.38382700085639954, "sampling/importance_sampling_ratio/mean": 1.0087472200393677, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.13137870281934738, "clip_ratio/low_mean": 0.005859375, "clip_ratio/low_min": 0.005859375, "clip_ratio/high_mean": 0.02217741869390011, "clip_ratio/high_max": 0.02217741869390011, "clip_ratio/region_mean": 0.02803679369390011, "reward_total_mean": 0.8264527320861816, "reward_meter_mean": 0.8264527320861816, "reward_meter_std": 0.1885237991809845, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8264527320861816, "reward_total_composite_std": 0.1885237991809845} {"timestamp_utc": "2026-04-11T23:23:11Z", "mode": "train", "global_step": 1002, "epoch": 0.040245812748523915, "loss": 0.0292, "grad_norm": 9.637553215026855, "learning_rate": 6.966666666666667e-06, "num_tokens": 2247834.0, "completions/mean_length": 65.875, "completions/min_length": 63.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.875, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.8884119987487793, "rewards/meter/std": 0.23295524716377258, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8884119987487793, "rewards/total_composite/std": 0.23295524716377258, "reward": 0.8884119987487793, "reward_std": 0.2329552173614502, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06217576563358307, "sampling/sampling_logp_difference/max": 1.78690767288208, "sampling/importance_sampling_ratio/min": 0.16747726500034332, "sampling/importance_sampling_ratio/mean": 1.0028692483901978, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23851481638848782, "clip_ratio/low_mean": 0.012928195297718048, "clip_ratio/low_min": 0.012928195297718048, "clip_ratio/high_mean": 0.03800246538594365, "clip_ratio/high_max": 0.03800246538594365, "clip_ratio/region_mean": 0.0509306606836617, "reward_total_mean": 0.8884119987487793, "reward_meter_mean": 0.8884119987487793, "reward_meter_std": 0.23295524716377258, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8884119987487793, "reward_total_composite_std": 0.23295524716377258} {"timestamp_utc": "2026-04-11T23:23:16Z", "mode": "train", "global_step": 1003, "epoch": 0.04028597823030887, "loss": 0.0045, "grad_norm": 5.540070056915283, "learning_rate": 6.963636363636364e-06, "num_tokens": 2250215.0, "completions/mean_length": 114.625, "completions/min_length": 112.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.625, "completions/min_terminated_length": 112.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.781272828578949, "rewards/meter/std": 0.39195698499679565, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8214285373687744, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.634056806564331, "rewards/total_composite/std": 0.31964385509490967, "reward": 0.634056806564331, "reward_std": 0.31964385509490967, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014822704717516899, "sampling/sampling_logp_difference/max": 1.5009106397628784, "sampling/importance_sampling_ratio/min": 0.22292707860469818, "sampling/importance_sampling_ratio/mean": 0.9998385906219482, "sampling/importance_sampling_ratio/max": 1.965267539024353, "entropy": 0.05840234411880374, "clip_ratio/low_mean": 0.004366895416751504, "clip_ratio/low_min": 0.004366895416751504, "clip_ratio/high_mean": 0.005464061861857772, "clip_ratio/high_max": 0.005464061861857772, "clip_ratio/region_mean": 0.009830957278609276, "reward_total_mean": 0.634056806564331, "reward_meter_mean": 0.781272828578949, "reward_meter_std": 0.39195698499679565, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8214285373687744, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.634056806564331, "reward_total_composite_std": 0.31964385509490967} {"timestamp_utc": "2026-04-11T23:23:21Z", "mode": "train", "global_step": 1004, "epoch": 0.04032614371209382, "loss": 0.0036, "grad_norm": 7.136742115020752, "learning_rate": 6.960606060606061e-06, "num_tokens": 2251916.0, "completions/mean_length": 59.625, "completions/min_length": 58.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.625, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9974349737167358, "rewards/meter/std": 0.0002844088012352586, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9974349737167358, "rewards/total_composite/std": 0.0002844088012352586, "reward": 0.9974349737167358, "reward_std": 0.0002844203554559499, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018237588927149773, "sampling/sampling_logp_difference/max": 1.3813776969909668, "sampling/importance_sampling_ratio/min": 0.25123220682144165, "sampling/importance_sampling_ratio/mean": 0.9971798062324524, "sampling/importance_sampling_ratio/max": 1.5021758079528809, "entropy": 0.06448704563081264, "clip_ratio/low_mean": 0.004273816477507353, "clip_ratio/low_min": 0.004273816477507353, "clip_ratio/high_mean": 0.014583333861082792, "clip_ratio/high_max": 0.014583333861082792, "clip_ratio/region_mean": 0.018857150338590145, "reward_total_mean": 0.9974349737167358, "reward_meter_mean": 0.9974349737167358, "reward_meter_std": 0.0002844088012352586, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9974349737167358, "reward_total_composite_std": 0.0002844088012352586} {"timestamp_utc": "2026-04-11T23:23:27Z", "mode": "train", "global_step": 1005, "epoch": 0.04036630919387878, "loss": 0.0226, "grad_norm": 2.455113649368286, "learning_rate": 6.957575757575759e-06, "num_tokens": 2255101.0, "completions/mean_length": 197.125, "completions/min_length": 190.0, "completions/max_length": 203.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 197.125, "completions/min_terminated_length": 190.0, "completions/max_terminated_length": 203.0, "rewards/meter/mean": 0.9832211136817932, "rewards/meter/std": 0.004388164728879929, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7840908765792847, "rewards/repeat_penalty/std": 0.06763852387666702, "rewards/total_composite/mean": 0.6168004274368286, "rewards/total_composite/std": 0.054140087217092514, "reward": 0.6168004274368286, "reward_std": 0.054140083491802216, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023318301886320114, "sampling/sampling_logp_difference/max": 1.2863155603408813, "sampling/importance_sampling_ratio/min": 0.27628687024116516, "sampling/importance_sampling_ratio/mean": 1.0025383234024048, "sampling/importance_sampling_ratio/max": 1.8359591960906982, "entropy": 0.11270508635789156, "clip_ratio/low_mean": 0.005632697488181293, "clip_ratio/low_min": 0.005632697488181293, "clip_ratio/high_mean": 0.014221564109902829, "clip_ratio/high_max": 0.014221564109902829, "clip_ratio/region_mean": 0.019854261598084122, "reward_total_mean": 0.6168004274368286, "reward_meter_mean": 0.9832211136817932, "reward_meter_std": 0.004388164728879929, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7840908765792847, "reward_repeat_penalty_std": 0.06763852387666702, "reward_total_composite_mean": 0.6168004274368286, "reward_total_composite_std": 0.054140087217092514} {"timestamp_utc": "2026-04-11T23:23:32Z", "mode": "train", "global_step": 1006, "epoch": 0.04040647467566374, "loss": -0.0394, "grad_norm": 7.012386798858643, "learning_rate": 6.954545454545455e-06, "num_tokens": 2257080.0, "completions/mean_length": 91.375, "completions/min_length": 80.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.375, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.9163761734962463, "rewards/meter/std": 0.1686118096113205, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7331008911132812, "rewards/total_composite/std": 0.13488943874835968, "reward": 0.7331008911132812, "reward_std": 0.1348894238471985, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024793440476059914, "sampling/sampling_logp_difference/max": 1.819629192352295, "sampling/importance_sampling_ratio/min": 0.16208584606647491, "sampling/importance_sampling_ratio/mean": 0.9956486225128174, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09509005583822727, "clip_ratio/low_mean": 0.0031250000465661287, "clip_ratio/low_min": 0.0031250000465661287, "clip_ratio/high_mean": 0.010740866768173873, "clip_ratio/high_max": 0.010740866768173873, "clip_ratio/region_mean": 0.013865866814740002, "reward_total_mean": 0.7331008911132812, "reward_meter_mean": 0.9163761734962463, "reward_meter_std": 0.1686118096113205, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7331008911132812, "reward_total_composite_std": 0.13488943874835968} {"timestamp_utc": "2026-04-11T23:23:37Z", "mode": "train", "global_step": 1007, "epoch": 0.04044664015744869, "loss": -0.0001, "grad_norm": 2.0443665981292725, "learning_rate": 6.951515151515153e-06, "num_tokens": 2258850.0, "completions/mean_length": 60.25, "completions/min_length": 60.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.25, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9974480867385864, "rewards/meter/std": 0.00012538061127997935, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9974480867385864, "rewards/total_composite/std": 0.00012538061127997935, "reward": 0.9974480867385864, "reward_std": 0.00012537713337223977, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013044199906289577, "sampling/sampling_logp_difference/max": 0.7166652679443359, "sampling/importance_sampling_ratio/min": 0.4883781373500824, "sampling/importance_sampling_ratio/mean": 0.9988527894020081, "sampling/importance_sampling_ratio/max": 1.6381621360778809, "entropy": 0.04919788311235607, "clip_ratio/low_mean": 0.004166666883975267, "clip_ratio/low_min": 0.004166666883975267, "clip_ratio/high_mean": 0.008266129298135638, "clip_ratio/high_max": 0.008266129298135638, "clip_ratio/region_mean": 0.012432796182110906, "reward_total_mean": 0.9974480867385864, "reward_meter_mean": 0.9974480867385864, "reward_meter_std": 0.00012538061127997935, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9974480867385864, "reward_total_composite_std": 0.00012538061127997935} {"timestamp_utc": "2026-04-11T23:23:46Z", "mode": "train", "global_step": 1008, "epoch": 0.040486805639233646, "loss": 0.0083, "grad_norm": 2.1715681552886963, "learning_rate": 6.948484848484849e-06, "num_tokens": 2263802.0, "completions/mean_length": 423.0, "completions/min_length": 336.0, "completions/max_length": 471.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 423.0, "completions/min_terminated_length": 336.0, "completions/max_terminated_length": 471.0, "rewards/meter/mean": 0.029939208179712296, "rewards/meter/std": 0.05622902140021324, "rewards/count_adherence/mean": 0.9107142686843872, "rewards/count_adherence/std": 0.0832117572426796, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.5447744727134705, "rewards/repeat_penalty/std": 0.19969096779823303, "rewards/total_composite/mean": 0.017225507646799088, "rewards/total_composite/std": 0.03621317073702812, "reward": 0.017225507646799088, "reward_std": 0.036213167011737823, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014329123310744762, "sampling/sampling_logp_difference/max": 2.782862424850464, "sampling/importance_sampling_ratio/min": 0.061861179769039154, "sampling/importance_sampling_ratio/mean": 1.0038172006607056, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08116073720157146, "clip_ratio/low_mean": 0.010143639519810677, "clip_ratio/low_min": 0.010143639519810677, "clip_ratio/high_mean": 0.003859857562929392, "clip_ratio/high_max": 0.003859857562929392, "clip_ratio/region_mean": 0.014003497082740068, "reward_total_mean": 0.017225507646799088, "reward_meter_mean": 0.029939208179712296, "reward_meter_std": 0.05622902140021324, "reward_count_adherence_mean": 0.9107142686843872, "reward_count_adherence_std": 0.0832117572426796, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.5447744727134705, "reward_repeat_penalty_std": 0.19969096779823303, "reward_total_composite_mean": 0.017225507646799088, "reward_total_composite_std": 0.03621317073702812} {"timestamp_utc": "2026-04-11T23:23:52Z", "mode": "train", "global_step": 1009, "epoch": 0.0405269711210186, "loss": 0.0182, "grad_norm": 6.43966007232666, "learning_rate": 6.945454545454546e-06, "num_tokens": 2265904.0, "completions/mean_length": 88.75, "completions/min_length": 88.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 88.75, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.9968655705451965, "rewards/meter/std": 0.0008970692288130522, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7749999761581421, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.7726212739944458, "rewards/total_composite/std": 0.07099877297878265, "reward": 0.7726212739944458, "reward_std": 0.07099878042936325, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006405611056834459, "sampling/sampling_logp_difference/max": 0.8346314430236816, "sampling/importance_sampling_ratio/min": 0.6927711367607117, "sampling/importance_sampling_ratio/mean": 1.0016840696334839, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.02403299231082201, "clip_ratio/low_mean": 0.0027173913549631834, "clip_ratio/low_min": 0.0027173913549631834, "clip_ratio/high_mean": 0.004261363763362169, "clip_ratio/high_max": 0.004261363763362169, "clip_ratio/region_mean": 0.006978755118325353, "reward_total_mean": 0.7726212739944458, "reward_meter_mean": 0.9968655705451965, "reward_meter_std": 0.0008970692288130522, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7749999761581421, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.7726212739944458, "reward_total_composite_std": 0.07099877297878265} {"timestamp_utc": "2026-04-11T23:23:56Z", "mode": "train", "global_step": 1010, "epoch": 0.040567136602803554, "loss": 0.0216, "grad_norm": 7.395992755889893, "learning_rate": 6.942424242424243e-06, "num_tokens": 2267676.0, "completions/mean_length": 69.5, "completions/min_length": 66.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.5, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9437720775604248, "rewards/meter/std": 0.10555455833673477, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9021741151809692, "rewards/total_composite/std": 0.14069606363773346, "reward": 0.9021741151809692, "reward_std": 0.14069606363773346, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0496804304420948, "sampling/sampling_logp_difference/max": 1.3186415433883667, "sampling/importance_sampling_ratio/min": 0.2674984335899353, "sampling/importance_sampling_ratio/mean": 1.0084481239318848, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.22077207453548908, "clip_ratio/low_mean": 0.01056338008493185, "clip_ratio/low_min": 0.01056338008493185, "clip_ratio/high_mean": 0.023642246145755053, "clip_ratio/high_max": 0.023642246145755053, "clip_ratio/region_mean": 0.0342056262306869, "reward_total_mean": 0.9021741151809692, "reward_meter_mean": 0.9437720775604248, "reward_meter_std": 0.10555455833673477, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.9021741151809692, "reward_total_composite_std": 0.14069606363773346} {"timestamp_utc": "2026-04-11T23:24:01Z", "mode": "train", "global_step": 1011, "epoch": 0.04060730208458851, "loss": 0.0142, "grad_norm": 5.829395771026611, "learning_rate": 6.93939393939394e-06, "num_tokens": 2269520.0, "completions/mean_length": 68.5, "completions/min_length": 63.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.5, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9850254058837891, "rewards/meter/std": 0.01548534631729126, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9850254058837891, "rewards/total_composite/std": 0.01548534631729126, "reward": 0.9850254058837891, "reward_std": 0.015485321171581745, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06408417969942093, "sampling/sampling_logp_difference/max": 1.6191701889038086, "sampling/importance_sampling_ratio/min": 0.1980629861354828, "sampling/importance_sampling_ratio/mean": 0.9924511313438416, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2325905654579401, "clip_ratio/low_mean": 0.018277311231940985, "clip_ratio/low_min": 0.018277311231940985, "clip_ratio/high_mean": 0.047181503381580114, "clip_ratio/high_max": 0.047181503381580114, "clip_ratio/region_mean": 0.0654588146135211, "reward_total_mean": 0.9850254058837891, "reward_meter_mean": 0.9850254058837891, "reward_meter_std": 0.01548534631729126, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9850254058837891, "reward_total_composite_std": 0.01548534631729126} {"timestamp_utc": "2026-04-11T23:24:06Z", "mode": "train", "global_step": 1012, "epoch": 0.04064746756637346, "loss": 0.0315, "grad_norm": 5.250216960906982, "learning_rate": 6.936363636363636e-06, "num_tokens": 2271718.0, "completions/mean_length": 102.75, "completions/min_length": 91.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.75, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.9896959066390991, "rewards/meter/std": 0.009748869575560093, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9896959066390991, "rewards/total_composite/std": 0.009748869575560093, "reward": 0.9896959066390991, "reward_std": 0.009748890995979309, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05327339842915535, "sampling/sampling_logp_difference/max": 1.0836901664733887, "sampling/importance_sampling_ratio/min": 0.33834466338157654, "sampling/importance_sampling_ratio/mean": 1.0056629180908203, "sampling/importance_sampling_ratio/max": 1.915394902229309, "entropy": 0.30692350305616856, "clip_ratio/low_mean": 0.01664550113491714, "clip_ratio/low_min": 0.01664550113491714, "clip_ratio/high_mean": 0.03817800944671035, "clip_ratio/high_max": 0.03817800944671035, "clip_ratio/region_mean": 0.05482351058162749, "reward_total_mean": 0.9896959066390991, "reward_meter_mean": 0.9896959066390991, "reward_meter_std": 0.009748869575560093, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9896959066390991, "reward_total_composite_std": 0.009748869575560093} {"timestamp_utc": "2026-04-11T23:24:12Z", "mode": "train", "global_step": 1013, "epoch": 0.040687633048158416, "loss": -0.0071, "grad_norm": 5.079646110534668, "learning_rate": 6.9333333333333344e-06, "num_tokens": 2273974.0, "completions/mean_length": 114.0, "completions/min_length": 104.0, "completions/max_length": 120.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.0, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.004421628080308437, "rewards/meter/std": 0.004275813698768616, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.0039320518262684345, "rewards/total_composite/std": 0.004037159029394388, "reward": 0.0039320518262684345, "reward_std": 0.0040371594950556755, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.035947732627391815, "sampling/sampling_logp_difference/max": 1.231058120727539, "sampling/importance_sampling_ratio/min": 0.2919834554195404, "sampling/importance_sampling_ratio/mean": 1.0108870267868042, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.25786815397441387, "clip_ratio/low_mean": 0.01979458425194025, "clip_ratio/low_min": 0.01979458425194025, "clip_ratio/high_mean": 0.00749256182461977, "clip_ratio/high_max": 0.00749256182461977, "clip_ratio/region_mean": 0.02728714607656002, "reward_total_mean": 0.0039320518262684345, "reward_meter_mean": 0.004421628080308437, "reward_meter_std": 0.004275813698768616, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.0039320518262684345, "reward_total_composite_std": 0.004037159029394388} {"timestamp_utc": "2026-04-11T23:24:17Z", "mode": "train", "global_step": 1014, "epoch": 0.04072779852994337, "loss": 0.006, "grad_norm": 3.9192073345184326, "learning_rate": 6.930303030303031e-06, "num_tokens": 2275905.0, "completions/mean_length": 74.375, "completions/min_length": 70.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.375, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9924027919769287, "rewards/meter/std": 0.005311480723321438, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9924027919769287, "rewards/total_composite/std": 0.005311480723321438, "reward": 0.9924027919769287, "reward_std": 0.005311472807079554, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024386199191212654, "sampling/sampling_logp_difference/max": 0.6070661544799805, "sampling/importance_sampling_ratio/min": 0.5449473261833191, "sampling/importance_sampling_ratio/mean": 1.0100598335266113, "sampling/importance_sampling_ratio/max": 1.5236577987670898, "entropy": 0.16714145429432392, "clip_ratio/low_mean": 0.005090707214549184, "clip_ratio/low_min": 0.005090707214549184, "clip_ratio/high_mean": 0.020599488052539527, "clip_ratio/high_max": 0.020599488052539527, "clip_ratio/region_mean": 0.02569019526708871, "reward_total_mean": 0.9924027919769287, "reward_meter_mean": 0.9924027919769287, "reward_meter_std": 0.005311480723321438, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9924027919769287, "reward_total_composite_std": 0.005311480723321438} {"timestamp_utc": "2026-04-11T23:24:21Z", "mode": "train", "global_step": 1015, "epoch": 0.04076796401172832, "loss": 0.0154, "grad_norm": 6.082520008087158, "learning_rate": 6.927272727272728e-06, "num_tokens": 2277383.0, "completions/mean_length": 33.75, "completions/min_length": 32.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.75, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9623132944107056, "rewards/meter/std": 0.02709752693772316, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9623132944107056, "rewards/total_composite/std": 0.02709752693772316, "reward": 0.9623132944107056, "reward_std": 0.027097532525658607, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03504392132163048, "sampling/sampling_logp_difference/max": 1.1891875267028809, "sampling/importance_sampling_ratio/min": 0.3405103385448456, "sampling/importance_sampling_ratio/mean": 1.01416015625, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.22182095050811768, "clip_ratio/low_mean": 0.007352941203862429, "clip_ratio/low_min": 0.007352941203862429, "clip_ratio/high_mean": 0.007247899193316698, "clip_ratio/high_max": 0.007247899193316698, "clip_ratio/region_mean": 0.014600840397179127, "reward_total_mean": 0.9623132944107056, "reward_meter_mean": 0.9623132944107056, "reward_meter_std": 0.02709752693772316, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9623132944107056, "reward_total_composite_std": 0.02709752693772316} {"timestamp_utc": "2026-04-11T23:24:31Z", "mode": "train", "global_step": 1016, "epoch": 0.04080812949351328, "loss": -0.1996, "grad_norm": 1.2893778085708618, "learning_rate": 6.9242424242424245e-06, "num_tokens": 2281936.0, "completions/mean_length": 404.125, "completions/min_length": 374.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 388.71429443359375, "completions/min_terminated_length": 374.0, "completions/max_terminated_length": 400.0, "rewards/meter/mean": 0.8926552534103394, "rewards/meter/std": 0.2755105197429657, "rewards/count_adherence/mean": 0.8083332777023315, "rewards/count_adherence/std": 0.27472931146621704, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.5423148274421692, "rewards/repeat_penalty/std": 0.26660624146461487, "rewards/total_composite/mean": 0.37017643451690674, "rewards/total_composite/std": 0.22237010300159454, "reward": 0.37017643451690674, "reward_std": 0.22237010300159454, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015432414598762989, "sampling/sampling_logp_difference/max": 3.34275484085083, "sampling/importance_sampling_ratio/min": 0.035339467227458954, "sampling/importance_sampling_ratio/mean": 1.002998948097229, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.04754339391365647, "clip_ratio/low_mean": 0.002274398400913924, "clip_ratio/low_min": 0.002274398400913924, "clip_ratio/high_mean": 0.00996917171869427, "clip_ratio/high_max": 0.00996917171869427, "clip_ratio/region_mean": 0.012243570119608194, "reward_total_mean": 0.37017643451690674, "reward_meter_mean": 0.8926552534103394, "reward_meter_std": 0.2755105197429657, "reward_count_adherence_mean": 0.8083332777023315, "reward_count_adherence_std": 0.27472931146621704, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.5423148274421692, "reward_repeat_penalty_std": 0.26660624146461487, "reward_total_composite_mean": 0.37017643451690674, "reward_total_composite_std": 0.22237010300159454} {"timestamp_utc": "2026-04-11T23:24:36Z", "mode": "train", "global_step": 1017, "epoch": 0.04084829497529823, "loss": -0.0018, "grad_norm": 4.289333343505859, "learning_rate": 6.921212121212122e-06, "num_tokens": 2283872.0, "completions/mean_length": 75.0, "completions/min_length": 73.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.0, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9958159327507019, "rewards/meter/std": 0.0007772990502417088, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9958159327507019, "rewards/total_composite/std": 0.0007772990502417088, "reward": 0.9958159327507019, "reward_std": 0.0007773060933686793, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05131428688764572, "sampling/sampling_logp_difference/max": 1.3117103576660156, "sampling/importance_sampling_ratio/min": 0.26935896277427673, "sampling/importance_sampling_ratio/mean": 1.0068724155426025, "sampling/importance_sampling_ratio/max": 1.9063295125961304, "entropy": 0.2947879508137703, "clip_ratio/low_mean": 0.013492224738001823, "clip_ratio/low_min": 0.013492224738001823, "clip_ratio/high_mean": 0.02297498215921223, "clip_ratio/high_max": 0.02297498215921223, "clip_ratio/region_mean": 0.036467206897214055, "reward_total_mean": 0.9958159327507019, "reward_meter_mean": 0.9958159327507019, "reward_meter_std": 0.0007772990502417088, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9958159327507019, "reward_total_composite_std": 0.0007772990502417088} {"timestamp_utc": "2026-04-11T23:24:41Z", "mode": "train", "global_step": 1018, "epoch": 0.040888460457083185, "loss": 0.0122, "grad_norm": 6.446331977844238, "learning_rate": 6.918181818181818e-06, "num_tokens": 2285685.0, "completions/mean_length": 69.625, "completions/min_length": 67.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.625, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9972782135009766, "rewards/meter/std": 0.0013893006835132837, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9972782135009766, "rewards/total_composite/std": 0.0013893006835132837, "reward": 0.9972782135009766, "reward_std": 0.0013893075520172715, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09435711801052094, "sampling/sampling_logp_difference/max": 6.117408752441406, "sampling/importance_sampling_ratio/min": 0.0022041599731892347, "sampling/importance_sampling_ratio/mean": 1.0028208494186401, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4668789766728878, "clip_ratio/low_mean": 0.02322303969413042, "clip_ratio/low_min": 0.02322303969413042, "clip_ratio/high_mean": 0.057335917837917805, "clip_ratio/high_max": 0.057335917837917805, "clip_ratio/region_mean": 0.08055895753204823, "reward_total_mean": 0.9972782135009766, "reward_meter_mean": 0.9972782135009766, "reward_meter_std": 0.0013893006835132837, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9972782135009766, "reward_total_composite_std": 0.0013893006835132837} {"timestamp_utc": "2026-04-11T23:24:46Z", "mode": "train", "global_step": 1019, "epoch": 0.04092862593886814, "loss": -0.0074, "grad_norm": 8.041248321533203, "learning_rate": 6.915151515151515e-06, "num_tokens": 2287596.0, "completions/mean_length": 82.875, "completions/min_length": 77.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.875, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.42031192779541016, "rewards/meter/std": 0.37385284900665283, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.42031192779541016, "rewards/total_composite/std": 0.37385284900665283, "reward": 0.42031192779541016, "reward_std": 0.37385284900665283, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04477884620428085, "sampling/sampling_logp_difference/max": 1.506969928741455, "sampling/importance_sampling_ratio/min": 0.221580371260643, "sampling/importance_sampling_ratio/mean": 0.9995813965797424, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2028534710407257, "clip_ratio/low_mean": 0.02015177276916802, "clip_ratio/low_min": 0.02015177276916802, "clip_ratio/high_mean": 0.023812503553926945, "clip_ratio/high_max": 0.023812503553926945, "clip_ratio/region_mean": 0.043964276323094964, "reward_total_mean": 0.42031192779541016, "reward_meter_mean": 0.42031192779541016, "reward_meter_std": 0.37385284900665283, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.42031192779541016, "reward_total_composite_std": 0.37385284900665283} {"timestamp_utc": "2026-04-11T23:24:51Z", "mode": "train", "global_step": 1020, "epoch": 0.04096879142065309, "loss": 0.0039, "grad_norm": 3.4062325954437256, "learning_rate": 6.912121212121212e-06, "num_tokens": 2290034.0, "completions/mean_length": 115.75, "completions/min_length": 109.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.75, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.9905360341072083, "rewards/meter/std": 0.011200688779354095, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.8668707013130188, "rewards/total_composite/std": 0.10446701943874359, "reward": 0.8668707013130188, "reward_std": 0.10446702688932419, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031655244529247284, "sampling/sampling_logp_difference/max": 1.1195735931396484, "sampling/importance_sampling_ratio/min": 0.3525734841823578, "sampling/importance_sampling_ratio/mean": 1.0034315586090088, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.20292328391224146, "clip_ratio/low_mean": 0.0065684852888807654, "clip_ratio/low_min": 0.0065684852888807654, "clip_ratio/high_mean": 0.01801690785214305, "clip_ratio/high_max": 0.01801690785214305, "clip_ratio/region_mean": 0.024585393141023815, "reward_total_mean": 0.8668707013130188, "reward_meter_mean": 0.9905360341072083, "reward_meter_std": 0.011200688779354095, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.8668707013130188, "reward_total_composite_std": 0.10446701943874359} {"timestamp_utc": "2026-04-11T23:24:56Z", "mode": "train", "global_step": 1021, "epoch": 0.04100895690243805, "loss": -0.0207, "grad_norm": 5.57641077041626, "learning_rate": 6.90909090909091e-06, "num_tokens": 2291779.0, "completions/mean_length": 69.125, "completions/min_length": 63.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.125, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.1570926308631897, "rewards/meter/std": 0.11832874268293381, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.1570926308631897, "rewards/total_composite/std": 0.11832874268293381, "reward": 0.1570926308631897, "reward_std": 0.1183287501335144, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.038187529891729355, "sampling/sampling_logp_difference/max": 0.8104653358459473, "sampling/importance_sampling_ratio/min": 0.44465112686157227, "sampling/importance_sampling_ratio/mean": 1.0030486583709717, "sampling/importance_sampling_ratio/max": 1.6118385791778564, "entropy": 0.2529278863221407, "clip_ratio/low_mean": 0.022817460587248206, "clip_ratio/low_min": 0.022817460587248206, "clip_ratio/high_mean": 0.023018090520054102, "clip_ratio/high_max": 0.023018090520054102, "clip_ratio/region_mean": 0.04583555110730231, "reward_total_mean": 0.1570926308631897, "reward_meter_mean": 0.1570926308631897, "reward_meter_std": 0.11832874268293381, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.1570926308631897, "reward_total_composite_std": 0.11832874268293381} {"timestamp_utc": "2026-04-11T23:25:01Z", "mode": "train", "global_step": 1022, "epoch": 0.041049122384223, "loss": 0.0374, "grad_norm": 3.1528494358062744, "learning_rate": 6.906060606060606e-06, "num_tokens": 2293742.0, "completions/mean_length": 82.375, "completions/min_length": 75.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.375, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.9961868524551392, "rewards/meter/std": 0.0018421995919197798, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9961868524551392, "rewards/total_composite/std": 0.0018421995919197798, "reward": 0.9961868524551392, "reward_std": 0.0018422063440084457, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.036886442452669144, "sampling/sampling_logp_difference/max": 1.0753283500671387, "sampling/importance_sampling_ratio/min": 0.34118568897247314, "sampling/importance_sampling_ratio/mean": 1.0033493041992188, "sampling/importance_sampling_ratio/max": 1.9688695669174194, "entropy": 0.21936733089387417, "clip_ratio/low_mean": 0.008588795084506273, "clip_ratio/low_min": 0.008588795084506273, "clip_ratio/high_mean": 0.02277943748049438, "clip_ratio/high_max": 0.02277943748049438, "clip_ratio/region_mean": 0.03136823256500065, "reward_total_mean": 0.9961868524551392, "reward_meter_mean": 0.9961868524551392, "reward_meter_std": 0.0018421995919197798, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9961868524551392, "reward_total_composite_std": 0.0018421995919197798} {"timestamp_utc": "2026-04-11T23:25:06Z", "mode": "train", "global_step": 1023, "epoch": 0.041089287866007955, "loss": 0.0171, "grad_norm": 5.055918216705322, "learning_rate": 6.903030303030304e-06, "num_tokens": 2295652.0, "completions/mean_length": 68.75, "completions/min_length": 66.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.75, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.2434961199760437, "rewards/meter/std": 0.1991458386182785, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.2434961199760437, "rewards/total_composite/std": 0.1991458386182785, "reward": 0.2434961199760437, "reward_std": 0.1991458386182785, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05065210163593292, "sampling/sampling_logp_difference/max": 1.4522520303726196, "sampling/importance_sampling_ratio/min": 0.23404262959957123, "sampling/importance_sampling_ratio/mean": 1.0007877349853516, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2538141794502735, "clip_ratio/low_mean": 0.019885645247995853, "clip_ratio/low_min": 0.019885645247995853, "clip_ratio/high_mean": 0.023702416568994522, "clip_ratio/high_max": 0.023702416568994522, "clip_ratio/region_mean": 0.043588061816990376, "reward_total_mean": 0.2434961199760437, "reward_meter_mean": 0.2434961199760437, "reward_meter_std": 0.1991458386182785, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.2434961199760437, "reward_total_composite_std": 0.1991458386182785} {"timestamp_utc": "2026-04-11T23:25:15Z", "mode": "train", "global_step": 1024, "epoch": 0.04112945334779291, "loss": -0.0145, "grad_norm": 0.9608837962150574, "learning_rate": 6.9e-06, "num_tokens": 2300311.0, "completions/mean_length": 388.375, "completions/min_length": 351.0, "completions/max_length": 406.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 388.375, "completions/min_terminated_length": 351.0, "completions/max_terminated_length": 406.0, "rewards/meter/mean": 0.988775372505188, "rewards/meter/std": 0.01991611160337925, "rewards/count_adherence/mean": 0.7142857313156128, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6315789222717285, "rewards/repeat_penalty/std": 0.05626552551984787, "rewards/total_composite/mean": 0.44603413343429565, "rewards/total_composite/std": 0.0405924916267395, "reward": 0.44603413343429565, "reward_std": 0.0405924953520298, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012534739449620247, "sampling/sampling_logp_difference/max": 1.506673812866211, "sampling/importance_sampling_ratio/min": 0.22164598107337952, "sampling/importance_sampling_ratio/mean": 1.0023077726364136, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08917687484063208, "clip_ratio/low_mean": 0.004947002336848527, "clip_ratio/low_min": 0.004947002336848527, "clip_ratio/high_mean": 0.006947804824449122, "clip_ratio/high_max": 0.006947804824449122, "clip_ratio/region_mean": 0.01189480716129765, "reward_total_mean": 0.44603413343429565, "reward_meter_mean": 0.988775372505188, "reward_meter_std": 0.01991611160337925, "reward_count_adherence_mean": 0.7142857313156128, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6315789222717285, "reward_repeat_penalty_std": 0.05626552551984787, "reward_total_composite_mean": 0.44603413343429565, "reward_total_composite_std": 0.0405924916267395} {"timestamp_utc": "2026-04-11T23:25:20Z", "mode": "train", "global_step": 1025, "epoch": 0.04116961882957786, "loss": 0.0116, "grad_norm": 4.088324069976807, "learning_rate": 6.896969696969697e-06, "num_tokens": 2302372.0, "completions/mean_length": 97.625, "completions/min_length": 92.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.625, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.6049685478210449, "rewards/meter/std": 0.33388465642929077, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.1511857807636261, "rewards/total_composite/mean": 0.4717435836791992, "rewards/total_composite/std": 0.2634972631931305, "reward": 0.4717435836791992, "reward_std": 0.2634972631931305, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03984498605132103, "sampling/sampling_logp_difference/max": 1.4417171478271484, "sampling/importance_sampling_ratio/min": 0.23652127385139465, "sampling/importance_sampling_ratio/mean": 1.010300636291504, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2695270273834467, "clip_ratio/low_mean": 0.01788918604142964, "clip_ratio/low_min": 0.01788918604142964, "clip_ratio/high_mean": 0.011576289543882012, "clip_ratio/high_max": 0.011576289543882012, "clip_ratio/region_mean": 0.02946547558531165, "reward_total_mean": 0.4717435836791992, "reward_meter_mean": 0.6049685478210449, "reward_meter_std": 0.33388465642929077, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.1511857807636261, "reward_total_composite_mean": 0.4717435836791992, "reward_total_composite_std": 0.2634972631931305} {"timestamp_utc": "2026-04-11T23:25:25Z", "mode": "train", "global_step": 1026, "epoch": 0.04120978431136282, "loss": -0.0083, "grad_norm": 4.054835796356201, "learning_rate": 6.893939393939395e-06, "num_tokens": 2304286.0, "completions/mean_length": 61.25, "completions/min_length": 60.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.25, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.990164577960968, "rewards/meter/std": 0.004598760511726141, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.990164577960968, "rewards/total_composite/std": 0.004598760511726141, "reward": 0.990164577960968, "reward_std": 0.004598761908710003, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02800033800303936, "sampling/sampling_logp_difference/max": 1.0160877704620361, "sampling/importance_sampling_ratio/min": 0.3620084226131439, "sampling/importance_sampling_ratio/mean": 1.0041474103927612, "sampling/importance_sampling_ratio/max": 1.7953885793685913, "entropy": 0.15393794886767864, "clip_ratio/low_mean": 0.020427904091775417, "clip_ratio/low_min": 0.020427904091775417, "clip_ratio/high_mean": 0.001953125, "clip_ratio/high_max": 0.001953125, "clip_ratio/region_mean": 0.022381029091775417, "reward_total_mean": 0.990164577960968, "reward_meter_mean": 0.990164577960968, "reward_meter_std": 0.004598760511726141, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.990164577960968, "reward_total_composite_std": 0.004598760511726141} {"timestamp_utc": "2026-04-11T23:25:30Z", "mode": "train", "global_step": 1027, "epoch": 0.04124994979314777, "loss": -0.0138, "grad_norm": 3.0363378524780273, "learning_rate": 6.890909090909092e-06, "num_tokens": 2306411.0, "completions/mean_length": 90.625, "completions/min_length": 89.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.625, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.9952998757362366, "rewards/meter/std": 0.0028919558972120285, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.8707464337348938, "rewards/total_composite/std": 0.10161302238702774, "reward": 0.8707464337348938, "reward_std": 0.10161302983760834, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030290186405181885, "sampling/sampling_logp_difference/max": 1.4388771057128906, "sampling/importance_sampling_ratio/min": 0.23719395697116852, "sampling/importance_sampling_ratio/mean": 0.9983088374137878, "sampling/importance_sampling_ratio/max": 1.4683204889297485, "entropy": 0.18594534322619438, "clip_ratio/low_mean": 0.012531211483292282, "clip_ratio/low_min": 0.012531211483292282, "clip_ratio/high_mean": 0.0187515887664631, "clip_ratio/high_max": 0.0187515887664631, "clip_ratio/region_mean": 0.03128280024975538, "reward_total_mean": 0.8707464337348938, "reward_meter_mean": 0.9952998757362366, "reward_meter_std": 0.0028919558972120285, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.8707464337348938, "reward_total_composite_std": 0.10161302238702774} {"timestamp_utc": "2026-04-11T23:25:34Z", "mode": "train", "global_step": 1028, "epoch": 0.041290115274932725, "loss": 0.0248, "grad_norm": 12.675787925720215, "learning_rate": 6.887878787878789e-06, "num_tokens": 2308182.0, "completions/mean_length": 66.375, "completions/min_length": 64.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.375, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9956432580947876, "rewards/meter/std": 0.003459326224401593, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9956432580947876, "rewards/total_composite/std": 0.003459326224401593, "reward": 0.9956432580947876, "reward_std": 0.003459327155724168, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08104093372821808, "sampling/sampling_logp_difference/max": 2.1169023513793945, "sampling/importance_sampling_ratio/min": 0.12040401995182037, "sampling/importance_sampling_ratio/mean": 1.0011348724365234, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39760861173272133, "clip_ratio/low_mean": 0.011456389911472797, "clip_ratio/low_min": 0.011456389911472797, "clip_ratio/high_mean": 0.06568031571805477, "clip_ratio/high_max": 0.06568031571805477, "clip_ratio/region_mean": 0.07713670562952757, "reward_total_mean": 0.9956432580947876, "reward_meter_mean": 0.9956432580947876, "reward_meter_std": 0.003459326224401593, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9956432580947876, "reward_total_composite_std": 0.003459326224401593} {"timestamp_utc": "2026-04-11T23:25:39Z", "mode": "train", "global_step": 1029, "epoch": 0.04133028075671768, "loss": 0.0126, "grad_norm": 5.34241247177124, "learning_rate": 6.8848484848484854e-06, "num_tokens": 2309806.0, "completions/mean_length": 47.0, "completions/min_length": 46.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.0, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.12917472422122955, "rewards/meter/std": 0.3293028473854065, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.12917472422122955, "rewards/total_composite/std": 0.3293028473854065, "reward": 0.12917472422122955, "reward_std": 0.3293028175830841, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07030355930328369, "sampling/sampling_logp_difference/max": 2.201859712600708, "sampling/importance_sampling_ratio/min": 0.11059728264808655, "sampling/importance_sampling_ratio/mean": 0.9958956837654114, "sampling/importance_sampling_ratio/max": 1.494503378868103, "entropy": 0.34705137088894844, "clip_ratio/low_mean": 0.026484928792342544, "clip_ratio/low_min": 0.026484928792342544, "clip_ratio/high_mean": 0.007978723384439945, "clip_ratio/high_max": 0.007978723384439945, "clip_ratio/region_mean": 0.03446365217678249, "reward_total_mean": 0.12917472422122955, "reward_meter_mean": 0.12917472422122955, "reward_meter_std": 0.3293028473854065, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.12917472422122955, "reward_total_composite_std": 0.3293028473854065} {"timestamp_utc": "2026-04-11T23:25:43Z", "mode": "train", "global_step": 1030, "epoch": 0.04137044623850263, "loss": -0.0076, "grad_norm": 8.219462394714355, "learning_rate": 6.881818181818183e-06, "num_tokens": 2311551.0, "completions/mean_length": 70.125, "completions/min_length": 67.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.125, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.12689465284347534, "rewards/meter/std": 0.10111062228679657, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.12689465284347534, "rewards/total_composite/std": 0.10111062228679657, "reward": 0.12689465284347534, "reward_std": 0.10111062228679657, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07496663182973862, "sampling/sampling_logp_difference/max": 1.6190156936645508, "sampling/importance_sampling_ratio/min": 0.19809359312057495, "sampling/importance_sampling_ratio/mean": 1.0122308731079102, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4643949382007122, "clip_ratio/low_mean": 0.02179320773575455, "clip_ratio/low_min": 0.02179320773575455, "clip_ratio/high_mean": 0.032955562230199575, "clip_ratio/high_max": 0.032955562230199575, "clip_ratio/region_mean": 0.054748769965954125, "reward_total_mean": 0.12689465284347534, "reward_meter_mean": 0.12689465284347534, "reward_meter_std": 0.10111062228679657, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.12689465284347534, "reward_total_composite_std": 0.10111062228679657} {"timestamp_utc": "2026-04-11T23:25:53Z", "mode": "train", "global_step": 1031, "epoch": 0.041410611720287586, "loss": -0.2286, "grad_norm": 1.1240724325180054, "learning_rate": 6.878787878787879e-06, "num_tokens": 2313782.0, "completions/mean_length": 224.875, "completions/min_length": 122.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 129.1666717529297, "completions/min_terminated_length": 122.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.6844993233680725, "rewards/meter/std": 0.36398857831954956, "rewards/count_adherence/mean": 0.8500000238418579, "rewards/count_adherence/std": 0.2777460217475891, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 0.8611111044883728, "rewards/repeat_penalty/std": 0.09848947077989578, "rewards/total_composite/mean": 0.5306993722915649, "rewards/total_composite/std": 0.33074015378952026, "reward": 0.5306993722915649, "reward_std": 0.33074015378952026, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03939226269721985, "sampling/sampling_logp_difference/max": 1.3999316692352295, "sampling/importance_sampling_ratio/min": 0.24661383032798767, "sampling/importance_sampling_ratio/mean": 1.0042911767959595, "sampling/importance_sampling_ratio/max": 1.6992144584655762, "entropy": 0.16609951201826334, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.027175352559424937, "clip_ratio/high_max": 0.027175352559424937, "clip_ratio/region_mean": 0.027175352559424937, "reward_total_mean": 0.5306993722915649, "reward_meter_mean": 0.6844993233680725, "reward_meter_std": 0.36398857831954956, "reward_count_adherence_mean": 0.8500000238418579, "reward_count_adherence_std": 0.2777460217475891, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 0.8611111044883728, "reward_repeat_penalty_std": 0.09848947077989578, "reward_total_composite_mean": 0.5306993722915649, "reward_total_composite_std": 0.33074015378952026} {"timestamp_utc": "2026-04-11T23:26:03Z", "mode": "train", "global_step": 1032, "epoch": 0.04145077720207254, "loss": -0.1879, "grad_norm": 1.4397751092910767, "learning_rate": 6.875757575757576e-06, "num_tokens": 2317508.0, "completions/mean_length": 322.75, "completions/min_length": 279.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 295.71429443359375, "completions/min_terminated_length": 279.0, "completions/max_terminated_length": 312.0, "rewards/meter/mean": 0.8027582168579102, "rewards/meter/std": 0.345333069562912, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.2070196568965912, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.630974292755127, "rewards/repeat_penalty/std": 0.19635772705078125, "rewards/total_composite/mean": 0.3974968194961548, "rewards/total_composite/std": 0.2327551245689392, "reward": 0.3974968194961548, "reward_std": 0.23275509476661682, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022777464240789413, "sampling/sampling_logp_difference/max": 3.0765089988708496, "sampling/importance_sampling_ratio/min": 0.04611998051404953, "sampling/importance_sampling_ratio/mean": 1.0034023523330688, "sampling/importance_sampling_ratio/max": 1.966111660003662, "entropy": 0.12301667220890522, "clip_ratio/low_mean": 0.005716652231058106, "clip_ratio/low_min": 0.005716652231058106, "clip_ratio/high_mean": 0.007109183439752087, "clip_ratio/high_max": 0.007109183439752087, "clip_ratio/region_mean": 0.012825835670810193, "reward_total_mean": 0.3974968194961548, "reward_meter_mean": 0.8027582168579102, "reward_meter_std": 0.345333069562912, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.2070196568965912, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.630974292755127, "reward_repeat_penalty_std": 0.19635772705078125, "reward_total_composite_mean": 0.3974968194961548, "reward_total_composite_std": 0.2327551245689392} {"timestamp_utc": "2026-04-11T23:26:07Z", "mode": "train", "global_step": 1033, "epoch": 0.041490942683857494, "loss": 0.0392, "grad_norm": 16.486671447753906, "learning_rate": 6.872727272727273e-06, "num_tokens": 2319195.0, "completions/mean_length": 50.875, "completions/min_length": 48.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.875, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.8368604779243469, "rewards/meter/std": 0.27409422397613525, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8368604779243469, "rewards/total_composite/std": 0.27409422397613525, "reward": 0.8368604779243469, "reward_std": 0.27409422397613525, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04724201560020447, "sampling/sampling_logp_difference/max": 1.2481775283813477, "sampling/importance_sampling_ratio/min": 0.28702741861343384, "sampling/importance_sampling_ratio/mean": 1.001241683959961, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.21698110736906528, "clip_ratio/low_mean": 0.002358490601181984, "clip_ratio/low_min": 0.002358490601181984, "clip_ratio/high_mean": 0.026730769779533148, "clip_ratio/high_max": 0.026730769779533148, "clip_ratio/region_mean": 0.029089260380715132, "reward_total_mean": 0.8368604779243469, "reward_meter_mean": 0.8368604779243469, "reward_meter_std": 0.27409422397613525, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8368604779243469, "reward_total_composite_std": 0.27409422397613525} {"timestamp_utc": "2026-04-11T23:26:12Z", "mode": "train", "global_step": 1034, "epoch": 0.04153110816564245, "loss": 0.003, "grad_norm": 5.985336780548096, "learning_rate": 6.869696969696971e-06, "num_tokens": 2321016.0, "completions/mean_length": 67.625, "completions/min_length": 63.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.625, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.053763557225465775, "rewards/meter/std": 0.056538958102464676, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.053763557225465775, "rewards/total_composite/std": 0.056538958102464676, "reward": 0.053763557225465775, "reward_std": 0.05653895437717438, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07932179421186447, "sampling/sampling_logp_difference/max": 1.892876386642456, "sampling/importance_sampling_ratio/min": 0.1506378948688507, "sampling/importance_sampling_ratio/mean": 1.0089935064315796, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5639077946543694, "clip_ratio/low_mean": 0.02950768917798996, "clip_ratio/low_min": 0.02950768917798996, "clip_ratio/high_mean": 0.022404347429983318, "clip_ratio/high_max": 0.022404347429983318, "clip_ratio/region_mean": 0.05191203660797328, "reward_total_mean": 0.053763557225465775, "reward_meter_mean": 0.053763557225465775, "reward_meter_std": 0.056538958102464676, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.053763557225465775, "reward_total_composite_std": 0.056538958102464676} {"timestamp_utc": "2026-04-11T23:26:17Z", "mode": "train", "global_step": 1035, "epoch": 0.0415712736474274, "loss": 0.0078, "grad_norm": 4.859825134277344, "learning_rate": 6.866666666666667e-06, "num_tokens": 2323151.0, "completions/mean_length": 95.875, "completions/min_length": 93.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.875, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.8145841360092163, "rewards/meter/std": 0.33935561776161194, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.7163466215133667, "rewards/total_composite/std": 0.304019570350647, "reward": 0.7163466215133667, "reward_std": 0.304019570350647, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08979999274015427, "sampling/sampling_logp_difference/max": 24.219942092895508, "sampling/importance_sampling_ratio/min": 3.0297920422528435e-11, "sampling/importance_sampling_ratio/mean": 0.9907196164131165, "sampling/importance_sampling_ratio/max": 1.6885980367660522, "entropy": 0.18867953680455685, "clip_ratio/low_mean": 0.006497558439150453, "clip_ratio/low_min": 0.006497558439150453, "clip_ratio/high_mean": 0.03407761920243502, "clip_ratio/high_max": 0.03407761920243502, "clip_ratio/region_mean": 0.04057517764158547, "reward_total_mean": 0.7163466215133667, "reward_meter_mean": 0.8145841360092163, "reward_meter_std": 0.33935561776161194, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.7163466215133667, "reward_total_composite_std": 0.304019570350647} {"timestamp_utc": "2026-04-11T23:26:21Z", "mode": "train", "global_step": 1036, "epoch": 0.041611439129212356, "loss": 0.0157, "grad_norm": 9.299543380737305, "learning_rate": 6.8636363636363645e-06, "num_tokens": 2324686.0, "completions/mean_length": 34.875, "completions/min_length": 33.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.875, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9909951686859131, "rewards/meter/std": 0.0066036987118422985, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9909951686859131, "rewards/total_composite/std": 0.0066036987118422985, "reward": 0.9909951686859131, "reward_std": 0.006603698246181011, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.047876570373773575, "sampling/sampling_logp_difference/max": 0.7040905952453613, "sampling/importance_sampling_ratio/min": 0.49455809593200684, "sampling/importance_sampling_ratio/mean": 1.0070772171020508, "sampling/importance_sampling_ratio/max": 1.8759939670562744, "entropy": 0.272716524079442, "clip_ratio/low_mean": 0.021446609403938055, "clip_ratio/low_min": 0.021446609403938055, "clip_ratio/high_mean": 0.021130952751263976, "clip_ratio/high_max": 0.021130952751263976, "clip_ratio/region_mean": 0.04257756215520203, "reward_total_mean": 0.9909951686859131, "reward_meter_mean": 0.9909951686859131, "reward_meter_std": 0.0066036987118422985, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9909951686859131, "reward_total_composite_std": 0.0066036987118422985} {"timestamp_utc": "2026-04-11T23:26:27Z", "mode": "train", "global_step": 1037, "epoch": 0.04165160461099731, "loss": -0.0406, "grad_norm": 13.22099494934082, "learning_rate": 6.860606060606061e-06, "num_tokens": 2327237.0, "completions/mean_length": 133.875, "completions/min_length": 109.0, "completions/max_length": 150.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 133.875, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 150.0, "rewards/meter/mean": 0.9935377836227417, "rewards/meter/std": 0.007430072408169508, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8857142925262451, "rewards/repeat_penalty/std": 0.07324227690696716, "rewards/total_composite/mean": 0.8550522923469543, "rewards/total_composite/std": 0.1225212812423706, "reward": 0.8550522923469543, "reward_std": 0.1225212812423706, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04958852007985115, "sampling/sampling_logp_difference/max": 2.24063777923584, "sampling/importance_sampling_ratio/min": 0.10639062523841858, "sampling/importance_sampling_ratio/mean": 1.0073775053024292, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2784906532615423, "clip_ratio/low_mean": 0.01997046614997089, "clip_ratio/low_min": 0.01997046614997089, "clip_ratio/high_mean": 0.018199163547251374, "clip_ratio/high_max": 0.018199163547251374, "clip_ratio/region_mean": 0.03816962969722226, "reward_total_mean": 0.8550522923469543, "reward_meter_mean": 0.9935377836227417, "reward_meter_std": 0.007430072408169508, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8857142925262451, "reward_repeat_penalty_std": 0.07324227690696716, "reward_total_composite_mean": 0.8550522923469543, "reward_total_composite_std": 0.1225212812423706} {"timestamp_utc": "2026-04-11T23:26:31Z", "mode": "train", "global_step": 1038, "epoch": 0.041691770092782264, "loss": 0.0376, "grad_norm": 9.27010726928711, "learning_rate": 6.857575757575758e-06, "num_tokens": 2328768.0, "completions/mean_length": 36.375, "completions/min_length": 33.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.375, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.24713511765003204, "rewards/meter/std": 0.3414146900177002, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.24713511765003204, "rewards/total_composite/std": 0.3414146900177002, "reward": 0.24713511765003204, "reward_std": 0.3414146900177002, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09440852701663971, "sampling/sampling_logp_difference/max": 1.0471820831298828, "sampling/importance_sampling_ratio/min": 0.3509252369403839, "sampling/importance_sampling_ratio/mean": 1.0138072967529297, "sampling/importance_sampling_ratio/max": 1.7845203876495361, "entropy": 0.8697552978992462, "clip_ratio/low_mean": 0.05519176600500941, "clip_ratio/low_min": 0.05519176600500941, "clip_ratio/high_mean": 0.017664092825725675, "clip_ratio/high_max": 0.017664092825725675, "clip_ratio/region_mean": 0.07285585883073509, "reward_total_mean": 0.24713511765003204, "reward_meter_mean": 0.24713511765003204, "reward_meter_std": 0.3414146900177002, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.24713511765003204, "reward_total_composite_std": 0.3414146900177002} {"timestamp_utc": "2026-04-11T23:26:36Z", "mode": "train", "global_step": 1039, "epoch": 0.04173193557456722, "loss": 0.0252, "grad_norm": 4.874062538146973, "learning_rate": 6.854545454545455e-06, "num_tokens": 2330669.0, "completions/mean_length": 69.625, "completions/min_length": 67.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.625, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.19316673278808594, "rewards/meter/std": 0.19894513487815857, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.19316673278808594, "rewards/total_composite/std": 0.19894513487815857, "reward": 0.19316673278808594, "reward_std": 0.19894511997699738, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0636194571852684, "sampling/sampling_logp_difference/max": 1.560652732849121, "sampling/importance_sampling_ratio/min": 0.2099989503622055, "sampling/importance_sampling_ratio/mean": 1.0039423704147339, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47740989923477173, "clip_ratio/low_mean": 0.03564694127999246, "clip_ratio/low_min": 0.03564694127999246, "clip_ratio/high_mean": 0.024171422701328993, "clip_ratio/high_max": 0.024171422701328993, "clip_ratio/region_mean": 0.059818363981321454, "reward_total_mean": 0.19316673278808594, "reward_meter_mean": 0.19316673278808594, "reward_meter_std": 0.19894513487815857, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.19316673278808594, "reward_total_composite_std": 0.19894513487815857} {"timestamp_utc": "2026-04-11T23:26:42Z", "mode": "train", "global_step": 1040, "epoch": 0.04177210105635217, "loss": -0.0641, "grad_norm": 5.022459030151367, "learning_rate": 6.851515151515153e-06, "num_tokens": 2333405.0, "completions/mean_length": 160.0, "completions/min_length": 131.0, "completions/max_length": 182.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 160.0, "completions/min_terminated_length": 131.0, "completions/max_terminated_length": 182.0, "rewards/meter/mean": 0.9006245136260986, "rewards/meter/std": 0.21294330060482025, "rewards/count_adherence/mean": 0.925000011920929, "rewards/count_adherence/std": 0.1035098284482956, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7440476417541504, "rewards/repeat_penalty/std": 0.17643596231937408, "rewards/total_composite/mean": 0.6102761030197144, "rewards/total_composite/std": 0.19687247276306152, "reward": 0.6102761030197144, "reward_std": 0.19687248766422272, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05596721172332764, "sampling/sampling_logp_difference/max": 1.3793213367462158, "sampling/importance_sampling_ratio/min": 0.25236961245536804, "sampling/importance_sampling_ratio/mean": 1.0185436010360718, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39750632736831903, "clip_ratio/low_mean": 0.016255452763289213, "clip_ratio/low_min": 0.016255452763289213, "clip_ratio/high_mean": 0.025824516778811812, "clip_ratio/high_max": 0.025824516778811812, "clip_ratio/region_mean": 0.042079969542101026, "reward_total_mean": 0.6102761030197144, "reward_meter_mean": 0.9006245136260986, "reward_meter_std": 0.21294330060482025, "reward_count_adherence_mean": 0.925000011920929, "reward_count_adherence_std": 0.1035098284482956, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7440476417541504, "reward_repeat_penalty_std": 0.17643596231937408, "reward_total_composite_mean": 0.6102761030197144, "reward_total_composite_std": 0.19687247276306152} {"timestamp_utc": "2026-04-11T23:26:46Z", "mode": "train", "global_step": 1041, "epoch": 0.041812266538137126, "loss": 0.0176, "grad_norm": 16.1435604095459, "learning_rate": 6.848484848484849e-06, "num_tokens": 2334958.0, "completions/mean_length": 40.125, "completions/min_length": 39.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.125, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.9823716878890991, "rewards/meter/std": 0.015363764949142933, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9823716878890991, "rewards/total_composite/std": 0.015363764949142933, "reward": 0.9823716878890991, "reward_std": 0.015363766811788082, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03611617907881737, "sampling/sampling_logp_difference/max": 1.3837116956710815, "sampling/importance_sampling_ratio/min": 0.2506465017795563, "sampling/importance_sampling_ratio/mean": 1.0097277164459229, "sampling/importance_sampling_ratio/max": 1.6625901460647583, "entropy": 0.2032633163034916, "clip_ratio/low_mean": 0.0031250000465661287, "clip_ratio/low_min": 0.0031250000465661287, "clip_ratio/high_mean": 0.012427689041942358, "clip_ratio/high_max": 0.012427689041942358, "clip_ratio/region_mean": 0.015552689088508487, "reward_total_mean": 0.9823716878890991, "reward_meter_mean": 0.9823716878890991, "reward_meter_std": 0.015363764949142933, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9823716878890991, "reward_total_composite_std": 0.015363764949142933} {"timestamp_utc": "2026-04-11T23:26:52Z", "mode": "train", "global_step": 1042, "epoch": 0.04185243201992208, "loss": 0.0581, "grad_norm": 3.8245151042938232, "learning_rate": 6.845454545454546e-06, "num_tokens": 2337338.0, "completions/mean_length": 126.5, "completions/min_length": 115.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 126.5, "completions/min_terminated_length": 115.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.7728732824325562, "rewards/meter/std": 0.3359214961528778, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.75, "rewards/repeat_penalty/std": 0.10101525485515594, "rewards/total_composite/mean": 0.5652507543563843, "rewards/total_composite/std": 0.22422268986701965, "reward": 0.5652507543563843, "reward_std": 0.22422267496585846, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031546179205179214, "sampling/sampling_logp_difference/max": 1.852052092552185, "sampling/importance_sampling_ratio/min": 0.1569148302078247, "sampling/importance_sampling_ratio/mean": 1.0046995878219604, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1586785214021802, "clip_ratio/low_mean": 0.007275132229551673, "clip_ratio/low_min": 0.007275132229551673, "clip_ratio/high_mean": 0.023097628843970597, "clip_ratio/high_max": 0.023097628843970597, "clip_ratio/region_mean": 0.03037276107352227, "reward_total_mean": 0.5652507543563843, "reward_meter_mean": 0.7728732824325562, "reward_meter_std": 0.3359214961528778, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.75, "reward_repeat_penalty_std": 0.10101525485515594, "reward_total_composite_mean": 0.5652507543563843, "reward_total_composite_std": 0.22422268986701965} {"timestamp_utc": "2026-04-11T23:26:57Z", "mode": "train", "global_step": 1043, "epoch": 0.041892597501707034, "loss": 0.0128, "grad_norm": 5.958546161651611, "learning_rate": 6.842424242424243e-06, "num_tokens": 2339119.0, "completions/mean_length": 67.625, "completions/min_length": 63.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.625, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.5403692722320557, "rewards/meter/std": 0.4386324882507324, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5403692722320557, "rewards/total_composite/std": 0.4386324882507324, "reward": 0.5403692722320557, "reward_std": 0.43863245844841003, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.047015149146318436, "sampling/sampling_logp_difference/max": 1.6448850631713867, "sampling/importance_sampling_ratio/min": 0.19303473830223083, "sampling/importance_sampling_ratio/mean": 1.0055934190750122, "sampling/importance_sampling_ratio/max": 1.7478914260864258, "entropy": 0.3001182395964861, "clip_ratio/low_mean": 0.02028265898115933, "clip_ratio/low_min": 0.02028265898115933, "clip_ratio/high_mean": 0.018411532044410706, "clip_ratio/high_max": 0.018411532044410706, "clip_ratio/region_mean": 0.038694191025570035, "reward_total_mean": 0.5403692722320557, "reward_meter_mean": 0.5403692722320557, "reward_meter_std": 0.4386324882507324, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5403692722320557, "reward_total_composite_std": 0.4386324882507324} {"timestamp_utc": "2026-04-11T23:27:02Z", "mode": "train", "global_step": 1044, "epoch": 0.04193276298349199, "loss": -0.0019, "grad_norm": 3.107640266418457, "learning_rate": 6.83939393939394e-06, "num_tokens": 2341333.0, "completions/mean_length": 98.75, "completions/min_length": 93.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.75, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.4054219126701355, "rewards/meter/std": 0.4075765311717987, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4054219126701355, "rewards/total_composite/std": 0.4075765311717987, "reward": 0.4054219126701355, "reward_std": 0.4075765013694763, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031202714890241623, "sampling/sampling_logp_difference/max": 0.8987345695495605, "sampling/importance_sampling_ratio/min": 0.40708449482917786, "sampling/importance_sampling_ratio/mean": 1.0041404962539673, "sampling/importance_sampling_ratio/max": 1.5978652238845825, "entropy": 0.1880979547277093, "clip_ratio/low_mean": 0.020476194215007126, "clip_ratio/low_min": 0.020476194215007126, "clip_ratio/high_mean": 0.006274061510339379, "clip_ratio/high_max": 0.006274061510339379, "clip_ratio/region_mean": 0.026750255725346506, "reward_total_mean": 0.4054219126701355, "reward_meter_mean": 0.4054219126701355, "reward_meter_std": 0.4075765311717987, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.4054219126701355, "reward_total_composite_std": 0.4075765311717987} {"timestamp_utc": "2026-04-11T23:27:07Z", "mode": "train", "global_step": 1045, "epoch": 0.04197292846527694, "loss": -0.0095, "grad_norm": 6.213621616363525, "learning_rate": 6.8363636363636364e-06, "num_tokens": 2343378.0, "completions/mean_length": 93.625, "completions/min_length": 89.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.625, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.745398998260498, "rewards/meter/std": 0.21442271769046783, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.745398998260498, "rewards/total_composite/std": 0.21442271769046783, "reward": 0.745398998260498, "reward_std": 0.21442271769046783, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.036333851516246796, "sampling/sampling_logp_difference/max": 1.0359835624694824, "sampling/importance_sampling_ratio/min": 0.35487717390060425, "sampling/importance_sampling_ratio/mean": 1.0055217742919922, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.20175535418093204, "clip_ratio/low_mean": 0.015083948150277138, "clip_ratio/low_min": 0.015083948150277138, "clip_ratio/high_mean": 0.021053530042991042, "clip_ratio/high_max": 0.021053530042991042, "clip_ratio/region_mean": 0.03613747819326818, "reward_total_mean": 0.745398998260498, "reward_meter_mean": 0.745398998260498, "reward_meter_std": 0.21442271769046783, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.745398998260498, "reward_total_composite_std": 0.21442271769046783} {"timestamp_utc": "2026-04-11T23:27:12Z", "mode": "train", "global_step": 1046, "epoch": 0.042013093947061896, "loss": 0.0232, "grad_norm": 8.28740406036377, "learning_rate": 6.833333333333334e-06, "num_tokens": 2345073.0, "completions/mean_length": 68.875, "completions/min_length": 66.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.875, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.5793697834014893, "rewards/meter/std": 0.473821759223938, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5793697834014893, "rewards/total_composite/std": 0.473821759223938, "reward": 0.5793697834014893, "reward_std": 0.473821759223938, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.058582451194524765, "sampling/sampling_logp_difference/max": 1.9549522399902344, "sampling/importance_sampling_ratio/min": 0.14157122373580933, "sampling/importance_sampling_ratio/mean": 1.0050064325332642, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3025083411484957, "clip_ratio/low_mean": 0.026830808725208044, "clip_ratio/low_min": 0.026830808725208044, "clip_ratio/high_mean": 0.031451608054339886, "clip_ratio/high_max": 0.031451608054339886, "clip_ratio/region_mean": 0.05828241677954793, "reward_total_mean": 0.5793697834014893, "reward_meter_mean": 0.5793697834014893, "reward_meter_std": 0.473821759223938, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5793697834014893, "reward_total_composite_std": 0.473821759223938} {"timestamp_utc": "2026-04-11T23:27:22Z", "mode": "train", "global_step": 1047, "epoch": 0.04205325942884685, "loss": -0.2118, "grad_norm": 1.4744713306427002, "learning_rate": 6.83030303030303e-06, "num_tokens": 2347417.0, "completions/mean_length": 181.0, "completions/min_length": 128.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 133.71429443359375, "completions/min_terminated_length": 128.0, "completions/max_terminated_length": 139.0, "rewards/meter/mean": 0.9518733024597168, "rewards/meter/std": 0.08209281414747238, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.7358262538909912, "rewards/total_composite/std": 0.31345105171203613, "reward": 0.7358262538909912, "reward_std": 0.31345105171203613, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05928022786974907, "sampling/sampling_logp_difference/max": 1.9391239881515503, "sampling/importance_sampling_ratio/min": 0.14382989704608917, "sampling/importance_sampling_ratio/mean": 1.0017281770706177, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.30693456530570984, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0447791040642187, "clip_ratio/high_max": 0.0447791040642187, "clip_ratio/region_mean": 0.0447791040642187, "reward_total_mean": 0.7358262538909912, "reward_meter_mean": 0.9518733024597168, "reward_meter_std": 0.08209281414747238, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.7358262538909912, "reward_total_composite_std": 0.31345105171203613} {"timestamp_utc": "2026-04-11T23:27:26Z", "mode": "train", "global_step": 1048, "epoch": 0.0420934249106318, "loss": -0.0116, "grad_norm": 8.561767578125, "learning_rate": 6.827272727272728e-06, "num_tokens": 2349002.0, "completions/mean_length": 47.125, "completions/min_length": 44.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.125, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.6955418586730957, "rewards/meter/std": 0.4201207160949707, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6955418586730957, "rewards/total_composite/std": 0.4201207160949707, "reward": 0.6955418586730957, "reward_std": 0.4201207160949707, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.051052458584308624, "sampling/sampling_logp_difference/max": 2.202850818634033, "sampling/importance_sampling_ratio/min": 0.11048772186040878, "sampling/importance_sampling_ratio/mean": 1.009395718574524, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.374115826562047, "clip_ratio/low_mean": 0.014204545877873898, "clip_ratio/low_min": 0.014204545877873898, "clip_ratio/high_mean": 0.026004510698840022, "clip_ratio/high_max": 0.026004510698840022, "clip_ratio/region_mean": 0.04020905657671392, "reward_total_mean": 0.6955418586730957, "reward_meter_mean": 0.6955418586730957, "reward_meter_std": 0.4201207160949707, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6955418586730957, "reward_total_composite_std": 0.4201207160949707} {"timestamp_utc": "2026-04-11T23:27:32Z", "mode": "train", "global_step": 1049, "epoch": 0.04213359039241676, "loss": 0.0138, "grad_norm": 8.349677085876465, "learning_rate": 6.824242424242425e-06, "num_tokens": 2351424.0, "completions/mean_length": 144.75, "completions/min_length": 139.0, "completions/max_length": 152.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 144.75, "completions/min_terminated_length": 139.0, "completions/max_terminated_length": 152.0, "rewards/meter/mean": 0.716080904006958, "rewards/meter/std": 0.3239102363586426, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.6298381090164185, "rewards/total_composite/std": 0.2847732603549957, "reward": 0.6298381090164185, "reward_std": 0.2847732603549957, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020672447979450226, "sampling/sampling_logp_difference/max": 1.2336063385009766, "sampling/importance_sampling_ratio/min": 0.2912403643131256, "sampling/importance_sampling_ratio/mean": 1.0015819072723389, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1037906464189291, "clip_ratio/low_mean": 0.012227151426486671, "clip_ratio/low_min": 0.012227151426486671, "clip_ratio/high_mean": 0.012093263794668019, "clip_ratio/high_max": 0.012093263794668019, "clip_ratio/region_mean": 0.02432041522115469, "reward_total_mean": 0.6298381090164185, "reward_meter_mean": 0.716080904006958, "reward_meter_std": 0.3239102363586426, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.6298381090164185, "reward_total_composite_std": 0.2847732603549957} {"timestamp_utc": "2026-04-11T23:27:37Z", "mode": "train", "global_step": 1050, "epoch": 0.04217375587420171, "loss": 0.0067, "grad_norm": 9.696529388427734, "learning_rate": 6.821212121212122e-06, "num_tokens": 2353303.0, "completions/mean_length": 72.875, "completions/min_length": 70.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.875, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.7290558815002441, "rewards/meter/std": 0.3074507713317871, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7290558815002441, "rewards/total_composite/std": 0.3074507713317871, "reward": 0.7290558815002441, "reward_std": 0.3074507415294647, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07516960054636002, "sampling/sampling_logp_difference/max": 3.8051562309265137, "sampling/importance_sampling_ratio/min": 0.02225572057068348, "sampling/importance_sampling_ratio/mean": 0.9999740719795227, "sampling/importance_sampling_ratio/max": 1.8488932847976685, "entropy": 0.3899807743728161, "clip_ratio/low_mean": 0.008656773250550032, "clip_ratio/low_min": 0.008656773250550032, "clip_ratio/high_mean": 0.020683051785454154, "clip_ratio/high_max": 0.020683051785454154, "clip_ratio/region_mean": 0.029339825036004186, "reward_total_mean": 0.7290558815002441, "reward_meter_mean": 0.7290558815002441, "reward_meter_std": 0.3074507713317871, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7290558815002441, "reward_total_composite_std": 0.3074507713317871} {"timestamp_utc": "2026-04-11T23:28:52Z", "mode": "eval", "global_step": 1050, "epoch": 0.04217375587420171, "eval_loss": NaN, "eval_runtime": 74.9052, "eval_samples_per_second": 1.388, "eval_steps_per_second": 0.174, "eval_num_tokens": 2353303.0, "eval_completions/mean_length": 226.30769230769232, "eval_completions/min_length": 73.84615384615384, "eval_completions/max_length": 395.0769230769231, "eval_completions/clipped_ratio": 0.009615384615384616, "eval_completions/mean_terminated_length": 223.2651108961839, "eval_completions/min_terminated_length": 73.84615384615384, "eval_completions/max_terminated_length": 383.3076923076923, "eval_rewards/meter/mean": 0.5502180938537304, "eval_rewards/meter/std": 0.40295018599583554, "eval_rewards/count_adherence/mean": 0.8679342178198007, "eval_rewards/count_adherence/std": 0.11943456530570984, "eval_rewards/arabic_clean/mean": 0.9807692307692307, "eval_rewards/arabic_clean/std": 0.05439282839114849, "eval_rewards/repeat_penalty/mean": 0.724188873401055, "eval_rewards/repeat_penalty/std": 0.2071983149418464, "eval_rewards/total_composite/mean": 0.33965545204969555, "eval_rewards/total_composite/std": 0.3024227091899285, "eval_reward": 0.33965545204969555, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.015186995840989627, "eval_sampling/sampling_logp_difference/max": 1.0244657076322115, "eval_sampling/importance_sampling_ratio/min": 0.3627954079554631, "eval_sampling/importance_sampling_ratio/mean": 1.0033980562136724, "eval_sampling/importance_sampling_ratio/max": 1.4160706263322096, "eval_entropy": 0.13593376485201028, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.33965545204969555, "eval_reward_meter_mean": 0.5502180938537304, "eval_reward_meter_std": 0.40295018599583554, "eval_reward_count_adherence_mean": 0.8679342178198007, "eval_reward_count_adherence_std": 0.11943456530570984, "eval_reward_arabic_clean_mean": 0.9807692307692307, "eval_reward_arabic_clean_std": 0.05439282839114849, "eval_reward_repeat_penalty_mean": 0.724188873401055, "eval_reward_repeat_penalty_std": 0.2071983149418464, "eval_reward_total_composite_mean": 0.33965545204969555, "eval_reward_total_composite_std": 0.3024227091899285} {"timestamp_utc": "2026-04-11T23:29:01Z", "mode": "train", "global_step": 1051, "epoch": 0.042213921355986665, "loss": 0.0139, "grad_norm": 2.1154820919036865, "learning_rate": 6.818181818181818e-06, "num_tokens": 2356570.0, "completions/mean_length": 212.375, "completions/min_length": 183.0, "completions/max_length": 235.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 212.375, "completions/min_terminated_length": 183.0, "completions/max_terminated_length": 235.0, "rewards/meter/mean": 0.802075982093811, "rewards/meter/std": 0.3460130989551544, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7899305820465088, "rewards/repeat_penalty/std": 0.090753473341465, "rewards/total_composite/mean": 0.5123095512390137, "rewards/total_composite/std": 0.2536679208278656, "reward": 0.5123095512390137, "reward_std": 0.2536678910255432, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.038277287036180496, "sampling/sampling_logp_difference/max": 1.5595903396606445, "sampling/importance_sampling_ratio/min": 0.21022215485572815, "sampling/importance_sampling_ratio/mean": 1.0035831928253174, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31745208986103535, "clip_ratio/low_mean": 0.005197861581109464, "clip_ratio/low_min": 0.005197861581109464, "clip_ratio/high_mean": 0.026332546956837177, "clip_ratio/high_max": 0.026332546956837177, "clip_ratio/region_mean": 0.03153040853794664, "reward_total_mean": 0.5123095512390137, "reward_meter_mean": 0.802075982093811, "reward_meter_std": 0.3460130989551544, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7899305820465088, "reward_repeat_penalty_std": 0.090753473341465, "reward_total_composite_mean": 0.5123095512390137, "reward_total_composite_std": 0.2536679208278656} {"timestamp_utc": "2026-04-11T23:29:06Z", "mode": "train", "global_step": 1052, "epoch": 0.04225408683777162, "loss": -0.0141, "grad_norm": 4.994956970214844, "learning_rate": 6.8151515151515155e-06, "num_tokens": 2358608.0, "completions/mean_length": 89.75, "completions/min_length": 84.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.75, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.9309325218200684, "rewards/meter/std": 0.17771661281585693, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9309325218200684, "rewards/total_composite/std": 0.17771661281585693, "reward": 0.9309325218200684, "reward_std": 0.17771658301353455, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07444296777248383, "sampling/sampling_logp_difference/max": 1.3812413215637207, "sampling/importance_sampling_ratio/min": 0.2512664496898651, "sampling/importance_sampling_ratio/mean": 1.0161553621292114, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.586686696857214, "clip_ratio/low_mean": 0.0028735632076859474, "clip_ratio/low_min": 0.0028735632076859474, "clip_ratio/high_mean": 0.05841635470278561, "clip_ratio/high_max": 0.05841635470278561, "clip_ratio/region_mean": 0.06128991791047156, "reward_total_mean": 0.9309325218200684, "reward_meter_mean": 0.9309325218200684, "reward_meter_std": 0.17771661281585693, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9309325218200684, "reward_total_composite_std": 0.17771661281585693} {"timestamp_utc": "2026-04-11T23:29:12Z", "mode": "train", "global_step": 1053, "epoch": 0.04229425231955657, "loss": 0.0063, "grad_norm": 4.2257232666015625, "learning_rate": 6.812121212121212e-06, "num_tokens": 2360466.0, "completions/mean_length": 69.25, "completions/min_length": 67.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.25, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9586122035980225, "rewards/meter/std": 0.00875465851277113, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9586122035980225, "rewards/total_composite/std": 0.00875465851277113, "reward": 0.9586122035980225, "reward_std": 0.00875465851277113, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020308073610067368, "sampling/sampling_logp_difference/max": 1.0632991790771484, "sampling/importance_sampling_ratio/min": 0.3453146815299988, "sampling/importance_sampling_ratio/mean": 0.9992114901542664, "sampling/importance_sampling_ratio/max": 1.3229237794876099, "entropy": 0.13288897648453712, "clip_ratio/low_mean": 0.01263241225387901, "clip_ratio/low_min": 0.01263241225387901, "clip_ratio/high_mean": 0.0037313431967049837, "clip_ratio/high_max": 0.0037313431967049837, "clip_ratio/region_mean": 0.016363755450583994, "reward_total_mean": 0.9586122035980225, "reward_meter_mean": 0.9586122035980225, "reward_meter_std": 0.00875465851277113, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9586122035980225, "reward_total_composite_std": 0.00875465851277113} {"timestamp_utc": "2026-04-11T23:29:17Z", "mode": "train", "global_step": 1054, "epoch": 0.04233441780134153, "loss": 0.0069, "grad_norm": 2.932860851287842, "learning_rate": 6.80909090909091e-06, "num_tokens": 2362481.0, "completions/mean_length": 96.875, "completions/min_length": 93.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.875, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.9400673508644104, "rewards/meter/std": 0.03255482763051987, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9002121686935425, "rewards/total_composite/std": 0.11076170206069946, "reward": 0.9002121686935425, "reward_std": 0.11076170206069946, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0384301133453846, "sampling/sampling_logp_difference/max": 1.6087217330932617, "sampling/importance_sampling_ratio/min": 0.20014327764511108, "sampling/importance_sampling_ratio/mean": 0.9991159439086914, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17869562469422817, "clip_ratio/low_mean": 0.008994319243356586, "clip_ratio/low_min": 0.008994319243356586, "clip_ratio/high_mean": 0.02460848237387836, "clip_ratio/high_max": 0.02460848237387836, "clip_ratio/region_mean": 0.033602801617234945, "reward_total_mean": 0.9002121686935425, "reward_meter_mean": 0.9400673508644104, "reward_meter_std": 0.03255482763051987, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.9002121686935425, "reward_total_composite_std": 0.11076170206069946} {"timestamp_utc": "2026-04-11T23:29:23Z", "mode": "train", "global_step": 1055, "epoch": 0.04237458328312648, "loss": 0.1006, "grad_norm": 3.7541143894195557, "learning_rate": 6.806060606060607e-06, "num_tokens": 2364800.0, "completions/mean_length": 137.875, "completions/min_length": 98.0, "completions/max_length": 156.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 137.875, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 156.0, "rewards/meter/mean": 0.25745514035224915, "rewards/meter/std": 0.2247081845998764, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.11785111576318741, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6964285969734192, "rewards/repeat_penalty/std": 0.2466983199119568, "rewards/total_composite/mean": 0.14893272519111633, "rewards/total_composite/std": 0.16499118506908417, "reward": 0.14893272519111633, "reward_std": 0.16499118506908417, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05157129094004631, "sampling/sampling_logp_difference/max": 1.8216896057128906, "sampling/importance_sampling_ratio/min": 0.16175222396850586, "sampling/importance_sampling_ratio/mean": 1.0185996294021606, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4246663171797991, "clip_ratio/low_mean": 0.014537818904500455, "clip_ratio/low_min": 0.014537818904500455, "clip_ratio/high_mean": 0.01200975279789418, "clip_ratio/high_max": 0.01200975279789418, "clip_ratio/region_mean": 0.026547571702394634, "reward_total_mean": 0.14893272519111633, "reward_meter_mean": 0.25745514035224915, "reward_meter_std": 0.2247081845998764, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.11785111576318741, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6964285969734192, "reward_repeat_penalty_std": 0.2466983199119568, "reward_total_composite_mean": 0.14893272519111633, "reward_total_composite_std": 0.16499118506908417} {"timestamp_utc": "2026-04-11T23:29:28Z", "mode": "train", "global_step": 1056, "epoch": 0.042414748764911435, "loss": 0.0447, "grad_norm": 12.402931213378906, "learning_rate": 6.803030303030304e-06, "num_tokens": 2366750.0, "completions/mean_length": 70.75, "completions/min_length": 68.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.75, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.30388349294662476, "rewards/meter/std": 0.2288217395544052, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.30388349294662476, "rewards/total_composite/std": 0.2288217395544052, "reward": 0.30388349294662476, "reward_std": 0.22882172465324402, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05324329808354378, "sampling/sampling_logp_difference/max": 2.614670753479004, "sampling/importance_sampling_ratio/min": 0.07319188117980957, "sampling/importance_sampling_ratio/mean": 0.9882052540779114, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17993433214724064, "clip_ratio/low_mean": 0.019135613925755024, "clip_ratio/low_min": 0.019135613925755024, "clip_ratio/high_mean": 0.02171629574149847, "clip_ratio/high_max": 0.02171629574149847, "clip_ratio/region_mean": 0.040851909667253494, "reward_total_mean": 0.30388349294662476, "reward_meter_mean": 0.30388349294662476, "reward_meter_std": 0.2288217395544052, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.30388349294662476, "reward_total_composite_std": 0.2288217395544052} {"timestamp_utc": "2026-04-11T23:29:37Z", "mode": "train", "global_step": 1057, "epoch": 0.04245491424669639, "loss": 0.0021, "grad_norm": 1.8229259252548218, "learning_rate": 6.800000000000001e-06, "num_tokens": 2371070.0, "completions/mean_length": 349.0, "completions/min_length": 337.0, "completions/max_length": 377.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 349.0, "completions/min_terminated_length": 337.0, "completions/max_terminated_length": 377.0, "rewards/meter/mean": 0.3180350661277771, "rewards/meter/std": 0.4220852255821228, "rewards/count_adherence/mean": 0.762499988079071, "rewards/count_adherence/std": 0.05175492912530899, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6007440090179443, "rewards/repeat_penalty/std": 0.11201495677232742, "rewards/total_composite/mean": 0.14668160676956177, "rewards/total_composite/std": 0.19722160696983337, "reward": 0.14668160676956177, "reward_std": 0.197221577167511, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020703818649053574, "sampling/sampling_logp_difference/max": 1.6899352073669434, "sampling/importance_sampling_ratio/min": 0.18453148007392883, "sampling/importance_sampling_ratio/mean": 1.001380205154419, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.14950765296816826, "clip_ratio/low_mean": 0.013276972225867212, "clip_ratio/low_min": 0.013276972225867212, "clip_ratio/high_mean": 0.0039429860189557076, "clip_ratio/high_max": 0.0039429860189557076, "clip_ratio/region_mean": 0.01721995824482292, "reward_total_mean": 0.14668160676956177, "reward_meter_mean": 0.3180350661277771, "reward_meter_std": 0.4220852255821228, "reward_count_adherence_mean": 0.762499988079071, "reward_count_adherence_std": 0.05175492912530899, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6007440090179443, "reward_repeat_penalty_std": 0.11201495677232742, "reward_total_composite_mean": 0.14668160676956177, "reward_total_composite_std": 0.19722160696983337} {"timestamp_utc": "2026-04-11T23:29:45Z", "mode": "train", "global_step": 1058, "epoch": 0.04249507972848134, "loss": -0.0193, "grad_norm": 2.157697916030884, "learning_rate": 6.796969696969697e-06, "num_tokens": 2375539.0, "completions/mean_length": 347.625, "completions/min_length": 326.0, "completions/max_length": 367.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 347.625, "completions/min_terminated_length": 326.0, "completions/max_terminated_length": 367.0, "rewards/meter/mean": 0.7610747814178467, "rewards/meter/std": 0.25479352474212646, "rewards/count_adherence/mean": 0.7222222089767456, "rewards/count_adherence/std": 0.059391383081674576, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4871794581413269, "rewards/repeat_penalty/std": 0.16510966420173645, "rewards/total_composite/mean": 0.26854199171066284, "rewards/total_composite/std": 0.14636212587356567, "reward": 0.26854199171066284, "reward_std": 0.14636212587356567, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017443938180804253, "sampling/sampling_logp_difference/max": 2.1108410358428955, "sampling/importance_sampling_ratio/min": 0.12113604694604874, "sampling/importance_sampling_ratio/mean": 1.0041615962982178, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07590485410764813, "clip_ratio/low_mean": 0.008206001890357584, "clip_ratio/low_min": 0.008206001890357584, "clip_ratio/high_mean": 0.0041363348718732595, "clip_ratio/high_max": 0.0041363348718732595, "clip_ratio/region_mean": 0.012342336762230843, "reward_total_mean": 0.26854199171066284, "reward_meter_mean": 0.7610747814178467, "reward_meter_std": 0.25479352474212646, "reward_count_adherence_mean": 0.7222222089767456, "reward_count_adherence_std": 0.059391383081674576, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4871794581413269, "reward_repeat_penalty_std": 0.16510966420173645, "reward_total_composite_mean": 0.26854199171066284, "reward_total_composite_std": 0.14636212587356567} {"timestamp_utc": "2026-04-11T23:29:50Z", "mode": "train", "global_step": 1059, "epoch": 0.0425352452102663, "loss": 0.0103, "grad_norm": 3.8445513248443604, "learning_rate": 6.793939393939395e-06, "num_tokens": 2377463.0, "completions/mean_length": 73.5, "completions/min_length": 71.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.5, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.1600387543439865, "rewards/meter/std": 0.3069753348827362, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.1600387543439865, "rewards/total_composite/std": 0.3069753348827362, "reward": 0.1600387543439865, "reward_std": 0.3069753646850586, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0510094054043293, "sampling/sampling_logp_difference/max": 1.574995994567871, "sampling/importance_sampling_ratio/min": 0.20700837671756744, "sampling/importance_sampling_ratio/mean": 1.008049726486206, "sampling/importance_sampling_ratio/max": 1.8910245895385742, "entropy": 0.32050950825214386, "clip_ratio/low_mean": 0.02205127221532166, "clip_ratio/low_min": 0.02205127221532166, "clip_ratio/high_mean": 0.006493506487458944, "clip_ratio/high_max": 0.006493506487458944, "clip_ratio/region_mean": 0.028544778702780604, "reward_total_mean": 0.1600387543439865, "reward_meter_mean": 0.1600387543439865, "reward_meter_std": 0.3069753348827362, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.1600387543439865, "reward_total_composite_std": 0.3069753348827362} {"timestamp_utc": "2026-04-11T23:30:00Z", "mode": "train", "global_step": 1060, "epoch": 0.04257541069205125, "loss": -0.3504, "grad_norm": 1.0387495756149292, "learning_rate": 6.790909090909091e-06, "num_tokens": 2381179.0, "completions/mean_length": 367.5, "completions/min_length": 299.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 319.3333435058594, "completions/min_terminated_length": 299.0, "completions/max_terminated_length": 347.0, "rewards/meter/mean": 0.8225868344306946, "rewards/meter/std": 0.18334050476551056, "rewards/count_adherence/mean": 0.7613636255264282, "rewards/count_adherence/std": 0.12798961997032166, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 0.7348039150238037, "rewards/repeat_penalty/std": 0.1173701137304306, "rewards/total_composite/mean": 0.3815041184425354, "rewards/total_composite/std": 0.23795181512832642, "reward": 0.3815041184425354, "reward_std": 0.23795180022716522, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028801359236240387, "sampling/sampling_logp_difference/max": 1.8230382204055786, "sampling/importance_sampling_ratio/min": 0.16153423488140106, "sampling/importance_sampling_ratio/mean": 1.0062330961227417, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1975250532850623, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.01580244250362739, "clip_ratio/high_max": 0.01580244250362739, "clip_ratio/region_mean": 0.01580244250362739, "reward_total_mean": 0.3815041184425354, "reward_meter_mean": 0.8225868344306946, "reward_meter_std": 0.18334050476551056, "reward_count_adherence_mean": 0.7613636255264282, "reward_count_adherence_std": 0.12798961997032166, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 0.7348039150238037, "reward_repeat_penalty_std": 0.1173701137304306, "reward_total_composite_mean": 0.3815041184425354, "reward_total_composite_std": 0.23795181512832642} {"timestamp_utc": "2026-04-11T23:30:07Z", "mode": "train", "global_step": 1061, "epoch": 0.042615576173836205, "loss": -0.0089, "grad_norm": 2.5813870429992676, "learning_rate": 6.787878787878789e-06, "num_tokens": 2385246.0, "completions/mean_length": 300.375, "completions/min_length": 289.0, "completions/max_length": 313.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 300.375, "completions/min_terminated_length": 289.0, "completions/max_terminated_length": 313.0, "rewards/meter/mean": 0.560438871383667, "rewards/meter/std": 0.4554891288280487, "rewards/count_adherence/mean": 0.8928571939468384, "rewards/count_adherence/std": 0.06613000482320786, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7440122365951538, "rewards/repeat_penalty/std": 0.10529463738203049, "rewards/total_composite/mean": 0.4012376070022583, "rewards/total_composite/std": 0.3423055410385132, "reward": 0.4012376070022583, "reward_std": 0.3423055112361908, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.032200075685977936, "sampling/sampling_logp_difference/max": 1.681164264678955, "sampling/importance_sampling_ratio/min": 0.18615710735321045, "sampling/importance_sampling_ratio/mean": 1.005553126335144, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.21212731674313545, "clip_ratio/low_mean": 0.015588806010782719, "clip_ratio/low_min": 0.015588806010782719, "clip_ratio/high_mean": 0.01467954705003649, "clip_ratio/high_max": 0.01467954705003649, "clip_ratio/region_mean": 0.03026835306081921, "reward_total_mean": 0.4012376070022583, "reward_meter_mean": 0.560438871383667, "reward_meter_std": 0.4554891288280487, "reward_count_adherence_mean": 0.8928571939468384, "reward_count_adherence_std": 0.06613000482320786, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7440122365951538, "reward_repeat_penalty_std": 0.10529463738203049, "reward_total_composite_mean": 0.4012376070022583, "reward_total_composite_std": 0.3423055410385132} {"timestamp_utc": "2026-04-11T23:30:14Z", "mode": "train", "global_step": 1062, "epoch": 0.04265574165562116, "loss": 0.0723, "grad_norm": 2.7553000450134277, "learning_rate": 6.7848484848484855e-06, "num_tokens": 2387921.0, "completions/mean_length": 177.375, "completions/min_length": 154.0, "completions/max_length": 207.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 177.375, "completions/min_terminated_length": 154.0, "completions/max_terminated_length": 207.0, "rewards/meter/mean": 0.3804289400577545, "rewards/meter/std": 0.4567885398864746, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5972222089767456, "rewards/repeat_penalty/std": 0.25845491886138916, "rewards/total_composite/mean": 0.1175825372338295, "rewards/total_composite/std": 0.16052623093128204, "reward": 0.1175825372338295, "reward_std": 0.16052624583244324, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02786877565085888, "sampling/sampling_logp_difference/max": 1.2727034091949463, "sampling/importance_sampling_ratio/min": 0.34907352924346924, "sampling/importance_sampling_ratio/mean": 1.005418062210083, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1391224949620664, "clip_ratio/low_mean": 0.016825482714921236, "clip_ratio/low_min": 0.016825482714921236, "clip_ratio/high_mean": 0.004816017230041325, "clip_ratio/high_max": 0.004816017230041325, "clip_ratio/region_mean": 0.02164149994496256, "reward_total_mean": 0.1175825372338295, "reward_meter_mean": 0.3804289400577545, "reward_meter_std": 0.4567885398864746, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5972222089767456, "reward_repeat_penalty_std": 0.25845491886138916, "reward_total_composite_mean": 0.1175825372338295, "reward_total_composite_std": 0.16052623093128204} {"timestamp_utc": "2026-04-11T23:30:19Z", "mode": "train", "global_step": 1063, "epoch": 0.04269590713740611, "loss": 0.0052, "grad_norm": 7.346792697906494, "learning_rate": 6.781818181818183e-06, "num_tokens": 2389705.0, "completions/mean_length": 66.0, "completions/min_length": 61.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.4735042452812195, "rewards/meter/std": 0.4612884223461151, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4735042452812195, "rewards/total_composite/std": 0.4612884223461151, "reward": 0.4735042452812195, "reward_std": 0.4612884521484375, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04212275519967079, "sampling/sampling_logp_difference/max": 1.0822772979736328, "sampling/importance_sampling_ratio/min": 0.3388230502605438, "sampling/importance_sampling_ratio/mean": 1.0069684982299805, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15965153090655804, "clip_ratio/low_mean": 0.014632937265560031, "clip_ratio/low_min": 0.014632937265560031, "clip_ratio/high_mean": 0.02170370938256383, "clip_ratio/high_max": 0.02170370938256383, "clip_ratio/region_mean": 0.03633664664812386, "reward_total_mean": 0.4735042452812195, "reward_meter_mean": 0.4735042452812195, "reward_meter_std": 0.4612884223461151, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.4735042452812195, "reward_total_composite_std": 0.4612884223461151} {"timestamp_utc": "2026-04-11T23:30:25Z", "mode": "train", "global_step": 1064, "epoch": 0.042736072619191066, "loss": 0.0144, "grad_norm": 2.248720407485962, "learning_rate": 6.778787878787879e-06, "num_tokens": 2392392.0, "completions/mean_length": 167.875, "completions/min_length": 163.0, "completions/max_length": 174.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 167.875, "completions/min_terminated_length": 163.0, "completions/max_terminated_length": 174.0, "rewards/meter/mean": 0.7978705167770386, "rewards/meter/std": 0.26100829243659973, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7222222089767456, "rewards/repeat_penalty/std": 0.059391383081674576, "rewards/total_composite/mean": 0.4253278374671936, "rewards/total_composite/std": 0.12737736105918884, "reward": 0.4253278374671936, "reward_std": 0.12737736105918884, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026998618617653847, "sampling/sampling_logp_difference/max": 4.974412441253662, "sampling/importance_sampling_ratio/min": 0.006912579294294119, "sampling/importance_sampling_ratio/mean": 1.0040228366851807, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15179980965331197, "clip_ratio/low_mean": 0.005747126415371895, "clip_ratio/low_min": 0.005747126415371895, "clip_ratio/high_mean": 0.002241053676698357, "clip_ratio/high_max": 0.002241053676698357, "clip_ratio/region_mean": 0.007988180092070252, "reward_total_mean": 0.4253278374671936, "reward_meter_mean": 0.7978705167770386, "reward_meter_std": 0.26100829243659973, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7222222089767456, "reward_repeat_penalty_std": 0.059391383081674576, "reward_total_composite_mean": 0.4253278374671936, "reward_total_composite_std": 0.12737736105918884} {"timestamp_utc": "2026-04-11T23:30:29Z", "mode": "train", "global_step": 1065, "epoch": 0.04277623810097602, "loss": 0.0031, "grad_norm": 7.699649810791016, "learning_rate": 6.7757575757575765e-06, "num_tokens": 2393979.0, "completions/mean_length": 51.375, "completions/min_length": 51.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.375, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.3520212769508362, "rewards/meter/std": 0.2575727701187134, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.3520212769508362, "rewards/total_composite/std": 0.2575727701187134, "reward": 0.3520212769508362, "reward_std": 0.2575727701187134, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03394267335534096, "sampling/sampling_logp_difference/max": 1.1649155616760254, "sampling/importance_sampling_ratio/min": 0.34641963243484497, "sampling/importance_sampling_ratio/mean": 1.0043641328811646, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17064102366566658, "clip_ratio/low_mean": 0.02187028736807406, "clip_ratio/low_min": 0.02187028736807406, "clip_ratio/high_mean": 0.014428413240239024, "clip_ratio/high_max": 0.014428413240239024, "clip_ratio/region_mean": 0.036298700608313084, "reward_total_mean": 0.3520212769508362, "reward_meter_mean": 0.3520212769508362, "reward_meter_std": 0.2575727701187134, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.3520212769508362, "reward_total_composite_std": 0.2575727701187134} {"timestamp_utc": "2026-04-11T23:30:34Z", "mode": "train", "global_step": 1066, "epoch": 0.042816403582760974, "loss": 0.0134, "grad_norm": 4.696002006530762, "learning_rate": 6.772727272727273e-06, "num_tokens": 2395725.0, "completions/mean_length": 67.25, "completions/min_length": 65.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.25, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.7684643864631653, "rewards/meter/std": 0.3259983956813812, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7684643864631653, "rewards/total_composite/std": 0.3259983956813812, "reward": 0.7684643864631653, "reward_std": 0.3259983956813812, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04000062495470047, "sampling/sampling_logp_difference/max": 1.5092440843582153, "sampling/importance_sampling_ratio/min": 0.22107702493667603, "sampling/importance_sampling_ratio/mean": 0.9985710978507996, "sampling/importance_sampling_ratio/max": 1.9387584924697876, "entropy": 0.23792543355375528, "clip_ratio/low_mean": 0.00916612590663135, "clip_ratio/low_min": 0.00916612590663135, "clip_ratio/high_mean": 0.022364818840287626, "clip_ratio/high_max": 0.022364818840287626, "clip_ratio/region_mean": 0.031530944746918976, "reward_total_mean": 0.7684643864631653, "reward_meter_mean": 0.7684643864631653, "reward_meter_std": 0.3259983956813812, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7684643864631653, "reward_total_composite_std": 0.3259983956813812} {"timestamp_utc": "2026-04-11T23:30:39Z", "mode": "train", "global_step": 1067, "epoch": 0.04285656906454593, "loss": 0.0273, "grad_norm": 10.04726791381836, "learning_rate": 6.76969696969697e-06, "num_tokens": 2397159.0, "completions/mean_length": 35.25, "completions/min_length": 33.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.25, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.40233293175697327, "rewards/meter/std": 0.35556212067604065, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.40233293175697327, "rewards/total_composite/std": 0.35556212067604065, "reward": 0.40233293175697327, "reward_std": 0.35556209087371826, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0461297444999218, "sampling/sampling_logp_difference/max": 1.1315374374389648, "sampling/importance_sampling_ratio/min": 0.32253700494766235, "sampling/importance_sampling_ratio/mean": 0.9973819255828857, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2819352522492409, "clip_ratio/low_mean": 0.010416666977107525, "clip_ratio/low_min": 0.010416666977107525, "clip_ratio/high_mean": 0.01831344375386834, "clip_ratio/high_max": 0.01831344375386834, "clip_ratio/region_mean": 0.028730110730975866, "reward_total_mean": 0.40233293175697327, "reward_meter_mean": 0.40233293175697327, "reward_meter_std": 0.35556212067604065, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.40233293175697327, "reward_total_composite_std": 0.35556212067604065} {"timestamp_utc": "2026-04-11T23:30:43Z", "mode": "train", "global_step": 1068, "epoch": 0.04289673454633088, "loss": 0.0118, "grad_norm": 3.635596752166748, "learning_rate": 6.7666666666666665e-06, "num_tokens": 2398762.0, "completions/mean_length": 47.375, "completions/min_length": 45.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.375, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.7638416290283203, "rewards/meter/std": 0.2724026143550873, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7638416290283203, "rewards/total_composite/std": 0.2724026143550873, "reward": 0.7638416290283203, "reward_std": 0.2724026143550873, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026672368869185448, "sampling/sampling_logp_difference/max": 1.1027817726135254, "sampling/importance_sampling_ratio/min": 0.33194637298583984, "sampling/importance_sampling_ratio/mean": 1.0070512294769287, "sampling/importance_sampling_ratio/max": 1.4794321060180664, "entropy": 0.15641734562814236, "clip_ratio/low_mean": 0.0076530613005161285, "clip_ratio/low_min": 0.0076530613005161285, "clip_ratio/high_mean": 0.02335231169126928, "clip_ratio/high_max": 0.02335231169126928, "clip_ratio/region_mean": 0.031005372991785407, "reward_total_mean": 0.7638416290283203, "reward_meter_mean": 0.7638416290283203, "reward_meter_std": 0.2724026143550873, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7638416290283203, "reward_total_composite_std": 0.2724026143550873} {"timestamp_utc": "2026-04-11T23:30:53Z", "mode": "train", "global_step": 1069, "epoch": 0.042936900028115836, "loss": -0.1635, "grad_norm": 1.4912827014923096, "learning_rate": 6.763636363636365e-06, "num_tokens": 2403084.0, "completions/mean_length": 370.25, "completions/min_length": 328.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 350.0000305175781, "completions/min_terminated_length": 328.0, "completions/max_terminated_length": 370.0, "rewards/meter/mean": 0.26017892360687256, "rewards/meter/std": 0.2312571257352829, "rewards/count_adherence/mean": 0.7291666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.6733911633491516, "rewards/repeat_penalty/std": 0.18306629359722137, "rewards/total_composite/mean": 0.10642074048519135, "rewards/total_composite/std": 0.06904294341802597, "reward": 0.10642074048519135, "reward_std": 0.06904293596744537, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02436882257461548, "sampling/sampling_logp_difference/max": 1.8471393585205078, "sampling/importance_sampling_ratio/min": 0.15768760442733765, "sampling/importance_sampling_ratio/mean": 1.0042506456375122, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11531824059784412, "clip_ratio/low_mean": 0.007008010521531105, "clip_ratio/low_min": 0.007008010521531105, "clip_ratio/high_mean": 0.009160695597529411, "clip_ratio/high_max": 0.009160695597529411, "clip_ratio/region_mean": 0.016168706119060516, "reward_total_mean": 0.10642074048519135, "reward_meter_mean": 0.26017892360687256, "reward_meter_std": 0.2312571257352829, "reward_count_adherence_mean": 0.7291666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.6733911633491516, "reward_repeat_penalty_std": 0.18306629359722137, "reward_total_composite_mean": 0.10642074048519135, "reward_total_composite_std": 0.06904294341802597} {"timestamp_utc": "2026-04-11T23:30:58Z", "mode": "train", "global_step": 1070, "epoch": 0.04297706550990079, "loss": 0.0141, "grad_norm": 7.106749534606934, "learning_rate": 6.760606060606061e-06, "num_tokens": 2404519.0, "completions/mean_length": 36.375, "completions/min_length": 34.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.375, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.9458709359169006, "rewards/meter/std": 0.06306823343038559, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9458709359169006, "rewards/total_composite/std": 0.06306823343038559, "reward": 0.9458709359169006, "reward_std": 0.0630682110786438, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.043692056089639664, "sampling/sampling_logp_difference/max": 1.2659907341003418, "sampling/importance_sampling_ratio/min": 0.28195980191230774, "sampling/importance_sampling_ratio/mean": 1.0076913833618164, "sampling/importance_sampling_ratio/max": 1.6269347667694092, "entropy": 0.26383113488554955, "clip_ratio/low_mean": 0.02384992502629757, "clip_ratio/low_min": 0.02384992502629757, "clip_ratio/high_mean": 0.006850600708276033, "clip_ratio/high_max": 0.006850600708276033, "clip_ratio/region_mean": 0.030700525734573603, "reward_total_mean": 0.9458709359169006, "reward_meter_mean": 0.9458709359169006, "reward_meter_std": 0.06306823343038559, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9458709359169006, "reward_total_composite_std": 0.06306823343038559} {"timestamp_utc": "2026-04-11T23:31:05Z", "mode": "train", "global_step": 1071, "epoch": 0.043017230991685744, "loss": 0.0028, "grad_norm": 1.8987226486206055, "learning_rate": 6.757575757575758e-06, "num_tokens": 2408287.0, "completions/mean_length": 257.0, "completions/min_length": 249.0, "completions/max_length": 262.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 257.0, "completions/min_terminated_length": 249.0, "completions/max_terminated_length": 262.0, "rewards/meter/mean": 0.8534860610961914, "rewards/meter/std": 0.08555062115192413, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6499999761581421, "rewards/repeat_penalty/std": 0.05909368395805359, "rewards/total_composite/mean": 0.4727606773376465, "rewards/total_composite/std": 0.03482650965452194, "reward": 0.4727606773376465, "reward_std": 0.03482651710510254, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011769634671509266, "sampling/sampling_logp_difference/max": 1.9065320491790771, "sampling/importance_sampling_ratio/min": 0.29330602288246155, "sampling/importance_sampling_ratio/mean": 1.001491665840149, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.04622745281085372, "clip_ratio/low_mean": 0.006310773896984756, "clip_ratio/low_min": 0.006310773896984756, "clip_ratio/high_mean": 0.0009578543831594288, "clip_ratio/high_max": 0.0009578543831594288, "clip_ratio/region_mean": 0.007268628280144185, "reward_total_mean": 0.4727606773376465, "reward_meter_mean": 0.8534860610961914, "reward_meter_std": 0.08555062115192413, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6499999761581421, "reward_repeat_penalty_std": 0.05909368395805359, "reward_total_composite_mean": 0.4727606773376465, "reward_total_composite_std": 0.03482650965452194} {"timestamp_utc": "2026-04-11T23:31:10Z", "mode": "train", "global_step": 1072, "epoch": 0.0430573964734707, "loss": -0.009, "grad_norm": 4.010406494140625, "learning_rate": 6.754545454545455e-06, "num_tokens": 2410412.0, "completions/mean_length": 96.625, "completions/min_length": 92.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.625, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.7025290727615356, "rewards/meter/std": 0.24330702424049377, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.6105661392211914, "rewards/total_composite/std": 0.21741104125976562, "reward": 0.6105661392211914, "reward_std": 0.21741104125976562, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02641892619431019, "sampling/sampling_logp_difference/max": 1.0504131317138672, "sampling/importance_sampling_ratio/min": 0.3497931957244873, "sampling/importance_sampling_ratio/mean": 1.0027921199798584, "sampling/importance_sampling_ratio/max": 1.5932246446609497, "entropy": 0.15043141692876816, "clip_ratio/low_mean": 0.003919956274330616, "clip_ratio/low_min": 0.003919956274330616, "clip_ratio/high_mean": 0.01675201370380819, "clip_ratio/high_max": 0.01675201370380819, "clip_ratio/region_mean": 0.020671969978138804, "reward_total_mean": 0.6105661392211914, "reward_meter_mean": 0.7025290727615356, "reward_meter_std": 0.24330702424049377, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.6105661392211914, "reward_total_composite_std": 0.21741104125976562} {"timestamp_utc": "2026-04-11T23:31:16Z", "mode": "train", "global_step": 1073, "epoch": 0.04309756195525565, "loss": -0.0119, "grad_norm": 4.956683158874512, "learning_rate": 6.751515151515152e-06, "num_tokens": 2412879.0, "completions/mean_length": 122.375, "completions/min_length": 117.0, "completions/max_length": 134.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.375, "completions/min_terminated_length": 117.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.8654184341430664, "rewards/meter/std": 0.3119131624698639, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.8155930042266846, "rewards/total_composite/std": 0.30145007371902466, "reward": 0.8155930042266846, "reward_std": 0.30145007371902466, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04843366891145706, "sampling/sampling_logp_difference/max": 1.2980079650878906, "sampling/importance_sampling_ratio/min": 0.2730752229690552, "sampling/importance_sampling_ratio/mean": 1.0030038356781006, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3304913900792599, "clip_ratio/low_mean": 0.012625776929780841, "clip_ratio/low_min": 0.012625776929780841, "clip_ratio/high_mean": 0.03231395175680518, "clip_ratio/high_max": 0.03231395175680518, "clip_ratio/region_mean": 0.04493972868658602, "reward_total_mean": 0.8155930042266846, "reward_meter_mean": 0.8654184341430664, "reward_meter_std": 0.3119131624698639, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.8155930042266846, "reward_total_composite_std": 0.30145007371902466} {"timestamp_utc": "2026-04-11T23:31:23Z", "mode": "train", "global_step": 1074, "epoch": 0.043137727437040606, "loss": 0.0024, "grad_norm": 2.035898208618164, "learning_rate": 6.748484848484848e-06, "num_tokens": 2416593.0, "completions/mean_length": 276.25, "completions/min_length": 261.0, "completions/max_length": 289.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 276.25, "completions/min_terminated_length": 261.0, "completions/max_terminated_length": 289.0, "rewards/meter/mean": 0.9123198986053467, "rewards/meter/std": 0.15345944464206696, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.08625821024179459, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6886509656906128, "rewards/repeat_penalty/std": 0.08202779293060303, "rewards/total_composite/mean": 0.5867777466773987, "rewards/total_composite/std": 0.12737217545509338, "reward": 0.5867777466773987, "reward_std": 0.12737217545509338, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01869548298418522, "sampling/sampling_logp_difference/max": 1.542731761932373, "sampling/importance_sampling_ratio/min": 0.21379627287387848, "sampling/importance_sampling_ratio/mean": 0.9994420409202576, "sampling/importance_sampling_ratio/max": 1.5070128440856934, "entropy": 0.1006718729622662, "clip_ratio/low_mean": 0.0050951789889950305, "clip_ratio/low_min": 0.0050951789889950305, "clip_ratio/high_mean": 0.0071886846562847495, "clip_ratio/high_max": 0.0071886846562847495, "clip_ratio/region_mean": 0.01228386364527978, "reward_total_mean": 0.5867777466773987, "reward_meter_mean": 0.9123198986053467, "reward_meter_std": 0.15345944464206696, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.08625821024179459, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6886509656906128, "reward_repeat_penalty_std": 0.08202779293060303, "reward_total_composite_mean": 0.5867777466773987, "reward_total_composite_std": 0.12737217545509338} {"timestamp_utc": "2026-04-11T23:31:31Z", "mode": "train", "global_step": 1075, "epoch": 0.04317789291882556, "loss": 0.012, "grad_norm": 1.7712578773498535, "learning_rate": 6.7454545454545465e-06, "num_tokens": 2420319.0, "completions/mean_length": 290.75, "completions/min_length": 268.0, "completions/max_length": 315.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 290.75, "completions/min_terminated_length": 268.0, "completions/max_terminated_length": 315.0, "rewards/meter/mean": 0.9649861454963684, "rewards/meter/std": 0.05583993345499039, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.634615421295166, "rewards/repeat_penalty/std": 0.03560846298933029, "rewards/total_composite/mean": 0.6113426089286804, "rewards/total_composite/std": 0.03276226669549942, "reward": 0.6113426089286804, "reward_std": 0.032762277871370316, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012581315822899342, "sampling/sampling_logp_difference/max": 1.3584985733032227, "sampling/importance_sampling_ratio/min": 0.2570464313030243, "sampling/importance_sampling_ratio/mean": 0.9999510049819946, "sampling/importance_sampling_ratio/max": 1.738738775253296, "entropy": 0.08378143096342683, "clip_ratio/low_mean": 0.0065517745388206095, "clip_ratio/low_min": 0.0065517745388206095, "clip_ratio/high_mean": 0.001742160296998918, "clip_ratio/high_max": 0.001742160296998918, "clip_ratio/region_mean": 0.008293934835819528, "reward_total_mean": 0.6113426089286804, "reward_meter_mean": 0.9649861454963684, "reward_meter_std": 0.05583993345499039, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.634615421295166, "reward_repeat_penalty_std": 0.03560846298933029, "reward_total_composite_mean": 0.6113426089286804, "reward_total_composite_std": 0.03276226669549942} {"timestamp_utc": "2026-04-11T23:31:37Z", "mode": "train", "global_step": 1076, "epoch": 0.043218058400610514, "loss": -0.0057, "grad_norm": 4.413632869720459, "learning_rate": 6.742424242424243e-06, "num_tokens": 2423119.0, "completions/mean_length": 170.0, "completions/min_length": 156.0, "completions/max_length": 186.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 170.0, "completions/min_terminated_length": 156.0, "completions/max_terminated_length": 186.0, "rewards/meter/mean": 0.7671191692352295, "rewards/meter/std": 0.2540895640850067, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7361111044883728, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.4233958125114441, "rewards/total_composite/std": 0.14756236970424652, "reward": 0.4233958125114441, "reward_std": 0.14756235480308533, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04489843547344208, "sampling/sampling_logp_difference/max": 8.813060760498047, "sampling/importance_sampling_ratio/min": 0.0001487771951360628, "sampling/importance_sampling_ratio/mean": 0.993308961391449, "sampling/importance_sampling_ratio/max": 1.7489451169967651, "entropy": 0.14643115364015102, "clip_ratio/low_mean": 0.012700201012194157, "clip_ratio/low_min": 0.012700201012194157, "clip_ratio/high_mean": 0.013822466367855668, "clip_ratio/high_max": 0.013822466367855668, "clip_ratio/region_mean": 0.026522667380049825, "reward_total_mean": 0.4233958125114441, "reward_meter_mean": 0.7671191692352295, "reward_meter_std": 0.2540895640850067, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7361111044883728, "reward_repeat_penalty_std": 0.05750546231865883, "reward_total_composite_mean": 0.4233958125114441, "reward_total_composite_std": 0.14756236970424652} {"timestamp_utc": "2026-04-11T23:31:41Z", "mode": "train", "global_step": 1077, "epoch": 0.04325822388239547, "loss": 0.012, "grad_norm": 3.8939266204833984, "learning_rate": 6.73939393939394e-06, "num_tokens": 2424867.0, "completions/mean_length": 57.5, "completions/min_length": 55.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.5, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9874081611633301, "rewards/meter/std": 0.004490252584218979, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6582721471786499, "rewards/total_composite/std": 0.0029935124330222607, "reward": 0.6582721471786499, "reward_std": 0.002993511501699686, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010735772550106049, "sampling/sampling_logp_difference/max": 0.34990429878234863, "sampling/importance_sampling_ratio/min": 0.7047555446624756, "sampling/importance_sampling_ratio/mean": 1.0056201219558716, "sampling/importance_sampling_ratio/max": 1.406072735786438, "entropy": 0.07559043914079666, "clip_ratio/low_mean": 0.0021186440717428923, "clip_ratio/low_min": 0.0021186440717428923, "clip_ratio/high_mean": 0.0065830720122903585, "clip_ratio/high_max": 0.0065830720122903585, "clip_ratio/region_mean": 0.00870171608403325, "reward_total_mean": 0.6582721471786499, "reward_meter_mean": 0.9874081611633301, "reward_meter_std": 0.004490252584218979, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6582721471786499, "reward_total_composite_std": 0.0029935124330222607} {"timestamp_utc": "2026-04-11T23:31:50Z", "mode": "train", "global_step": 1078, "epoch": 0.04329838936418042, "loss": -0.0117, "grad_norm": 0.8265510201454163, "learning_rate": 6.7363636363636365e-06, "num_tokens": 2429621.0, "completions/mean_length": 403.25, "completions/min_length": 384.0, "completions/max_length": 433.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 403.25, "completions/min_terminated_length": 384.0, "completions/max_terminated_length": 433.0, "rewards/meter/mean": 0.9796521067619324, "rewards/meter/std": 0.0062021855264902115, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5921052694320679, "rewards/repeat_penalty/std": 0.037216152995824814, "rewards/total_composite/mean": 0.38671886920928955, "rewards/total_composite/std": 0.02471131458878517, "reward": 0.38671886920928955, "reward_std": 0.024711309000849724, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005612175911664963, "sampling/sampling_logp_difference/max": 1.5524296760559082, "sampling/importance_sampling_ratio/min": 0.2117329090833664, "sampling/importance_sampling_ratio/mean": 1.0002822875976562, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03162188641726971, "clip_ratio/low_mean": 0.002776339795673266, "clip_ratio/low_min": 0.002776339795673266, "clip_ratio/high_mean": 0.0006038647261448205, "clip_ratio/high_max": 0.0006038647261448205, "clip_ratio/region_mean": 0.0033802045218180865, "reward_total_mean": 0.38671886920928955, "reward_meter_mean": 0.9796521067619324, "reward_meter_std": 0.0062021855264902115, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5921052694320679, "reward_repeat_penalty_std": 0.037216152995824814, "reward_total_composite_mean": 0.38671886920928955, "reward_total_composite_std": 0.02471131458878517} {"timestamp_utc": "2026-04-11T23:31:55Z", "mode": "train", "global_step": 1079, "epoch": 0.043338554845965375, "loss": 0.0067, "grad_norm": 4.684656143188477, "learning_rate": 6.733333333333334e-06, "num_tokens": 2431573.0, "completions/mean_length": 84.0, "completions/min_length": 81.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.0, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.9432131052017212, "rewards/meter/std": 0.005848580971360207, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9432131052017212, "rewards/total_composite/std": 0.005848580971360207, "reward": 0.9432131052017212, "reward_std": 0.0058485823683440685, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00942947156727314, "sampling/sampling_logp_difference/max": 1.2586157321929932, "sampling/importance_sampling_ratio/min": 0.2840469479560852, "sampling/importance_sampling_ratio/mean": 0.9998735785484314, "sampling/importance_sampling_ratio/max": 1.180408239364624, "entropy": 0.04214879055507481, "clip_ratio/low_mean": 0.0014367816038429737, "clip_ratio/low_min": 0.0014367816038429737, "clip_ratio/high_mean": 0.005959982983767986, "clip_ratio/high_max": 0.005959982983767986, "clip_ratio/region_mean": 0.00739676458761096, "reward_total_mean": 0.9432131052017212, "reward_meter_mean": 0.9432131052017212, "reward_meter_std": 0.005848580971360207, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9432131052017212, "reward_total_composite_std": 0.005848580971360207} {"timestamp_utc": "2026-04-11T23:32:00Z", "mode": "train", "global_step": 1080, "epoch": 0.04337872032775033, "loss": -0.088, "grad_norm": 7.663031578063965, "learning_rate": 6.73030303030303e-06, "num_tokens": 2433443.0, "completions/mean_length": 57.75, "completions/min_length": 51.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.75, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.31589365005493164, "rewards/meter/std": 0.41653454303741455, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.31589365005493164, "rewards/total_composite/std": 0.41653454303741455, "reward": 0.31589365005493164, "reward_std": 0.41653454303741455, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.038866013288497925, "sampling/sampling_logp_difference/max": 1.4967455863952637, "sampling/importance_sampling_ratio/min": 0.22385750710964203, "sampling/importance_sampling_ratio/mean": 1.0007740259170532, "sampling/importance_sampling_ratio/max": 1.726487159729004, "entropy": 0.14285273849964142, "clip_ratio/low_mean": 0.022774446289986372, "clip_ratio/low_min": 0.022774446289986372, "clip_ratio/high_mean": 0.014718614984303713, "clip_ratio/high_max": 0.014718614984303713, "clip_ratio/region_mean": 0.037493061274290085, "reward_total_mean": 0.31589365005493164, "reward_meter_mean": 0.31589365005493164, "reward_meter_std": 0.41653454303741455, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.31589365005493164, "reward_total_composite_std": 0.41653454303741455} {"timestamp_utc": "2026-04-11T23:32:05Z", "mode": "train", "global_step": 1081, "epoch": 0.04341888580953528, "loss": 0.0001, "grad_norm": 2.3848185539245605, "learning_rate": 6.7272727272727275e-06, "num_tokens": 2435373.0, "completions/mean_length": 84.25, "completions/min_length": 83.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.25, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.9907380938529968, "rewards/meter/std": 0.0005972905782982707, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8250000476837158, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8173696994781494, "rewards/total_composite/std": 0.07022644579410553, "reward": 0.8173696994781494, "reward_std": 0.07022644579410553, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018476024270057678, "sampling/sampling_logp_difference/max": 0.8972263336181641, "sampling/importance_sampling_ratio/min": 0.40769892930984497, "sampling/importance_sampling_ratio/mean": 0.9998801946640015, "sampling/importance_sampling_ratio/max": 1.8647558689117432, "entropy": 0.06444978155195713, "clip_ratio/low_mean": 0.010382481734268367, "clip_ratio/low_min": 0.010382481734268367, "clip_ratio/high_mean": 0.0029411765281111, "clip_ratio/high_max": 0.0029411765281111, "clip_ratio/region_mean": 0.013323658262379467, "reward_total_mean": 0.8173696994781494, "reward_meter_mean": 0.9907380938529968, "reward_meter_std": 0.0005972905782982707, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8250000476837158, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.8173696994781494, "reward_total_composite_std": 0.07022644579410553} {"timestamp_utc": "2026-04-11T23:32:10Z", "mode": "train", "global_step": 1082, "epoch": 0.04345905129132024, "loss": -0.0216, "grad_norm": 3.134387969970703, "learning_rate": 6.724242424242424e-06, "num_tokens": 2437378.0, "completions/mean_length": 77.625, "completions/min_length": 72.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.625, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.9952170252799988, "rewards/meter/std": 0.005060417577624321, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9952170252799988, "rewards/total_composite/std": 0.005060417577624321, "reward": 0.9952170252799988, "reward_std": 0.005060393828898668, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023099998012185097, "sampling/sampling_logp_difference/max": 1.7406330108642578, "sampling/importance_sampling_ratio/min": 0.17540933191776276, "sampling/importance_sampling_ratio/mean": 1.0053282976150513, "sampling/importance_sampling_ratio/max": 1.727461814880371, "entropy": 0.11937720235437155, "clip_ratio/low_mean": 0.0017361111240461469, "clip_ratio/low_min": 0.0017361111240461469, "clip_ratio/high_mean": 0.009705353993922472, "clip_ratio/high_max": 0.009705353993922472, "clip_ratio/region_mean": 0.011441465117968619, "reward_total_mean": 0.9952170252799988, "reward_meter_mean": 0.9952170252799988, "reward_meter_std": 0.005060417577624321, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9952170252799988, "reward_total_composite_std": 0.005060417577624321} {"timestamp_utc": "2026-04-11T23:32:16Z", "mode": "train", "global_step": 1083, "epoch": 0.04349921677310519, "loss": -0.0062, "grad_norm": 5.954423427581787, "learning_rate": 6.721212121212122e-06, "num_tokens": 2439776.0, "completions/mean_length": 127.75, "completions/min_length": 122.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.75, "completions/min_terminated_length": 122.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.7971546053886414, "rewards/meter/std": 0.2483818531036377, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7361111640930176, "rewards/repeat_penalty/std": 0.13197055459022522, "rewards/total_composite/mean": 0.43424269556999207, "rewards/total_composite/std": 0.14866070449352264, "reward": 0.43424269556999207, "reward_std": 0.14866070449352264, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02712804079055786, "sampling/sampling_logp_difference/max": 1.9316282272338867, "sampling/importance_sampling_ratio/min": 0.14491204917430878, "sampling/importance_sampling_ratio/mean": 1.0028880834579468, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10621183086186647, "clip_ratio/low_mean": 0.0030412437627092004, "clip_ratio/low_min": 0.0030412437627092004, "clip_ratio/high_mean": 0.01804604707285762, "clip_ratio/high_max": 0.01804604707285762, "clip_ratio/region_mean": 0.02108729083556682, "reward_total_mean": 0.43424269556999207, "reward_meter_mean": 0.7971546053886414, "reward_meter_std": 0.2483818531036377, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7361111640930176, "reward_repeat_penalty_std": 0.13197055459022522, "reward_total_composite_mean": 0.43424269556999207, "reward_total_composite_std": 0.14866070449352264} {"timestamp_utc": "2026-04-11T23:32:21Z", "mode": "train", "global_step": 1084, "epoch": 0.043539382254890145, "loss": 0.005, "grad_norm": 7.175000190734863, "learning_rate": 6.718181818181819e-06, "num_tokens": 2441759.0, "completions/mean_length": 80.875, "completions/min_length": 77.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.875, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.9879165291786194, "rewards/meter/std": 0.011460079811513424, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9879165291786194, "rewards/total_composite/std": 0.011460079811513424, "reward": 0.9879165291786194, "reward_std": 0.011460077948868275, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02571716159582138, "sampling/sampling_logp_difference/max": 1.5991357564926147, "sampling/importance_sampling_ratio/min": 0.20207108557224274, "sampling/importance_sampling_ratio/mean": 0.9987949132919312, "sampling/importance_sampling_ratio/max": 1.9328581094741821, "entropy": 0.10348887462168932, "clip_ratio/low_mean": 0.009149131015874445, "clip_ratio/low_min": 0.009149131015874445, "clip_ratio/high_mean": 0.018721832893788815, "clip_ratio/high_max": 0.018721832893788815, "clip_ratio/region_mean": 0.02787096390966326, "reward_total_mean": 0.9879165291786194, "reward_meter_mean": 0.9879165291786194, "reward_meter_std": 0.011460079811513424, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9879165291786194, "reward_total_composite_std": 0.011460079811513424} {"timestamp_utc": "2026-04-11T23:32:30Z", "mode": "train", "global_step": 1085, "epoch": 0.0435795477366751, "loss": -0.03, "grad_norm": 0.6841692924499512, "learning_rate": 6.715151515151516e-06, "num_tokens": 2446044.0, "completions/mean_length": 335.625, "completions/min_length": 318.0, "completions/max_length": 355.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 335.625, "completions/min_terminated_length": 318.0, "completions/max_terminated_length": 355.0, "rewards/meter/mean": 0.9889903664588928, "rewards/meter/std": 0.017724184319376945, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.05750546231865883, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5962929129600525, "rewards/repeat_penalty/std": 0.030314533039927483, "rewards/total_composite/mean": 0.5644394755363464, "rewards/total_composite/std": 0.03291511908173561, "reward": 0.5644394755363464, "reward_std": 0.03291511535644531, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008025525137782097, "sampling/sampling_logp_difference/max": 1.0485243797302246, "sampling/importance_sampling_ratio/min": 0.3504544794559479, "sampling/importance_sampling_ratio/mean": 0.9996162056922913, "sampling/importance_sampling_ratio/max": 1.9612756967544556, "entropy": 0.034420924610458314, "clip_ratio/low_mean": 0.0023487245198339224, "clip_ratio/low_min": 0.0023487245198339224, "clip_ratio/high_mean": 0.004307354771299288, "clip_ratio/high_max": 0.004307354771299288, "clip_ratio/region_mean": 0.00665607929113321, "reward_total_mean": 0.5644394755363464, "reward_meter_mean": 0.9889903664588928, "reward_meter_std": 0.017724184319376945, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.05750546231865883, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5962929129600525, "reward_repeat_penalty_std": 0.030314533039927483, "reward_total_composite_mean": 0.5644394755363464, "reward_total_composite_std": 0.03291511908173561} {"timestamp_utc": "2026-04-11T23:32:39Z", "mode": "train", "global_step": 1086, "epoch": 0.04361971321846005, "loss": -0.0088, "grad_norm": 0.8804117441177368, "learning_rate": 6.712121212121213e-06, "num_tokens": 2450552.0, "completions/mean_length": 350.5, "completions/min_length": 344.0, "completions/max_length": 369.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 350.5, "completions/min_terminated_length": 344.0, "completions/max_terminated_length": 369.0, "rewards/meter/mean": 0.9969491958618164, "rewards/meter/std": 0.0016905681695789099, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5882353186607361, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5277966260910034, "rewards/total_composite/std": 0.0008950114133767784, "reward": 0.5277966260910034, "reward_std": 0.00089502043556422, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005859900265932083, "sampling/sampling_logp_difference/max": 1.0715160369873047, "sampling/importance_sampling_ratio/min": 0.3424888849258423, "sampling/importance_sampling_ratio/mean": 0.999923050403595, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.025596523424610496, "clip_ratio/low_mean": 0.0010880097979679704, "clip_ratio/low_min": 0.0010880097979679704, "clip_ratio/high_mean": 0.0027567246870603412, "clip_ratio/high_max": 0.0027567246870603412, "clip_ratio/region_mean": 0.0038447344850283116, "reward_total_mean": 0.5277966260910034, "reward_meter_mean": 0.9969491958618164, "reward_meter_std": 0.0016905681695789099, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5882353186607361, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5277966260910034, "reward_total_composite_std": 0.0008950114133767784} {"timestamp_utc": "2026-04-11T23:32:44Z", "mode": "train", "global_step": 1087, "epoch": 0.04365987870024501, "loss": -0.0062, "grad_norm": 3.9621143341064453, "learning_rate": 6.709090909090909e-06, "num_tokens": 2452523.0, "completions/mean_length": 79.375, "completions/min_length": 74.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.375, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.7115691900253296, "rewards/meter/std": 0.38112178444862366, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7115691900253296, "rewards/total_composite/std": 0.38112178444862366, "reward": 0.7115691900253296, "reward_std": 0.38112178444862366, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.034753963351249695, "sampling/sampling_logp_difference/max": 2.0210351943969727, "sampling/importance_sampling_ratio/min": 0.1325182169675827, "sampling/importance_sampling_ratio/mean": 1.0051707029342651, "sampling/importance_sampling_ratio/max": 1.846152663230896, "entropy": 0.2076272815465927, "clip_ratio/low_mean": 0.006756756920367479, "clip_ratio/low_min": 0.006756756920367479, "clip_ratio/high_mean": 0.01255706837400794, "clip_ratio/high_max": 0.01255706837400794, "clip_ratio/region_mean": 0.01931382529437542, "reward_total_mean": 0.7115691900253296, "reward_meter_mean": 0.7115691900253296, "reward_meter_std": 0.38112178444862366, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7115691900253296, "reward_total_composite_std": 0.38112178444862366} {"timestamp_utc": "2026-04-11T23:32:48Z", "mode": "train", "global_step": 1088, "epoch": 0.04370004418202996, "loss": -0.0029, "grad_norm": 6.578573226928711, "learning_rate": 6.706060606060607e-06, "num_tokens": 2454071.0, "completions/mean_length": 33.5, "completions/min_length": 32.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.5, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.7622437477111816, "rewards/meter/std": 0.3664873540401459, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7622437477111816, "rewards/total_composite/std": 0.3664873540401459, "reward": 0.7622437477111816, "reward_std": 0.3664873540401459, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.040684185922145844, "sampling/sampling_logp_difference/max": 1.069122314453125, "sampling/importance_sampling_ratio/min": 0.3433097004890442, "sampling/importance_sampling_ratio/mean": 1.0093433856964111, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.18665132857859135, "clip_ratio/low_mean": 0.011363636702299118, "clip_ratio/low_min": 0.011363636702299118, "clip_ratio/high_mean": 0.04101851023733616, "clip_ratio/high_max": 0.04101851023733616, "clip_ratio/region_mean": 0.05238214693963528, "reward_total_mean": 0.7622437477111816, "reward_meter_mean": 0.7622437477111816, "reward_meter_std": 0.3664873540401459, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7622437477111816, "reward_total_composite_std": 0.3664873540401459} {"timestamp_utc": "2026-04-11T23:32:53Z", "mode": "train", "global_step": 1089, "epoch": 0.043740209663814915, "loss": 0.0411, "grad_norm": 4.875842571258545, "learning_rate": 6.703030303030304e-06, "num_tokens": 2455804.0, "completions/mean_length": 58.625, "completions/min_length": 54.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.625, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.8143894672393799, "rewards/meter/std": 0.22199998795986176, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8143894672393799, "rewards/total_composite/std": 0.22199998795986176, "reward": 0.8143894672393799, "reward_std": 0.22199997305870056, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04701100289821625, "sampling/sampling_logp_difference/max": 1.6473197937011719, "sampling/importance_sampling_ratio/min": 0.19256533682346344, "sampling/importance_sampling_ratio/mean": 0.9986065030097961, "sampling/importance_sampling_ratio/max": 1.836875081062317, "entropy": 0.1829883987084031, "clip_ratio/low_mean": 0.009895833441987634, "clip_ratio/low_min": 0.009895833441987634, "clip_ratio/high_mean": 0.03449019626714289, "clip_ratio/high_max": 0.03449019626714289, "clip_ratio/region_mean": 0.044386029709130526, "reward_total_mean": 0.8143894672393799, "reward_meter_mean": 0.8143894672393799, "reward_meter_std": 0.22199998795986176, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8143894672393799, "reward_total_composite_std": 0.22199998795986176} {"timestamp_utc": "2026-04-11T23:33:00Z", "mode": "train", "global_step": 1090, "epoch": 0.04378037514559987, "loss": -0.0213, "grad_norm": 1.876051902770996, "learning_rate": 6.700000000000001e-06, "num_tokens": 2459750.0, "completions/mean_length": 271.25, "completions/min_length": 258.0, "completions/max_length": 288.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 271.25, "completions/min_terminated_length": 258.0, "completions/max_terminated_length": 288.0, "rewards/meter/mean": 0.9913715124130249, "rewards/meter/std": 0.014886174350976944, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5326797366142273, "rewards/repeat_penalty/std": 0.2576701045036316, "rewards/total_composite/mean": 0.5263446569442749, "rewards/total_composite/std": 0.2508489787578583, "reward": 0.5263446569442749, "reward_std": 0.2508489787578583, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019940270110964775, "sampling/sampling_logp_difference/max": 1.2876369953155518, "sampling/importance_sampling_ratio/min": 0.275922030210495, "sampling/importance_sampling_ratio/mean": 1.0003299713134766, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10771761322394013, "clip_ratio/low_mean": 0.004208970727631822, "clip_ratio/low_min": 0.004208970727631822, "clip_ratio/high_mean": 0.012134980701375753, "clip_ratio/high_max": 0.012134980701375753, "clip_ratio/region_mean": 0.016343951429007575, "reward_total_mean": 0.5263446569442749, "reward_meter_mean": 0.9913715124130249, "reward_meter_std": 0.014886174350976944, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5326797366142273, "reward_repeat_penalty_std": 0.2576701045036316, "reward_total_composite_mean": 0.5263446569442749, "reward_total_composite_std": 0.2508489787578583} {"timestamp_utc": "2026-04-11T23:33:05Z", "mode": "train", "global_step": 1091, "epoch": 0.04382054062738482, "loss": -0.03, "grad_norm": 6.454976558685303, "learning_rate": 6.6969696969696975e-06, "num_tokens": 2461500.0, "completions/mean_length": 64.75, "completions/min_length": 57.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.75, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.5685036182403564, "rewards/meter/std": 0.32084769010543823, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5685036182403564, "rewards/total_composite/std": 0.32084769010543823, "reward": 0.5685036182403564, "reward_std": 0.32084769010543823, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.046030234545469284, "sampling/sampling_logp_difference/max": 1.455164909362793, "sampling/importance_sampling_ratio/min": 0.2333618849515915, "sampling/importance_sampling_ratio/mean": 1.007980227470398, "sampling/importance_sampling_ratio/max": 1.7851393222808838, "entropy": 0.27551514096558094, "clip_ratio/low_mean": 0.021997954230755568, "clip_ratio/low_min": 0.021997954230755568, "clip_ratio/high_mean": 0.005542142200283706, "clip_ratio/high_max": 0.005542142200283706, "clip_ratio/region_mean": 0.027540096431039274, "reward_total_mean": 0.5685036182403564, "reward_meter_mean": 0.5685036182403564, "reward_meter_std": 0.32084769010543823, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5685036182403564, "reward_total_composite_std": 0.32084769010543823} {"timestamp_utc": "2026-04-11T23:33:09Z", "mode": "train", "global_step": 1092, "epoch": 0.04386070610916978, "loss": 0.0095, "grad_norm": 4.461249351501465, "learning_rate": 6.693939393939395e-06, "num_tokens": 2463457.0, "completions/mean_length": 79.625, "completions/min_length": 75.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.625, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.8799495697021484, "rewards/meter/std": 0.1550467163324356, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8799495697021484, "rewards/total_composite/std": 0.1550467163324356, "reward": 0.8799495697021484, "reward_std": 0.1550467312335968, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021693911403417587, "sampling/sampling_logp_difference/max": 2.3354640007019043, "sampling/importance_sampling_ratio/min": 0.09676557034254074, "sampling/importance_sampling_ratio/mean": 1.0008015632629395, "sampling/importance_sampling_ratio/max": 1.76300048828125, "entropy": 0.07495173905044794, "clip_ratio/low_mean": 0.0030868902103975415, "clip_ratio/low_min": 0.0030868902103975415, "clip_ratio/high_mean": 0.006234930478967726, "clip_ratio/high_max": 0.006234930478967726, "clip_ratio/region_mean": 0.009321820689365268, "reward_total_mean": 0.8799495697021484, "reward_meter_mean": 0.8799495697021484, "reward_meter_std": 0.1550467163324356, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8799495697021484, "reward_total_composite_std": 0.1550467163324356} {"timestamp_utc": "2026-04-11T23:33:14Z", "mode": "train", "global_step": 1093, "epoch": 0.04390087159095473, "loss": 0.0004, "grad_norm": 10.553791046142578, "learning_rate": 6.690909090909091e-06, "num_tokens": 2465181.0, "completions/mean_length": 62.5, "completions/min_length": 60.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.5, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.8936691284179688, "rewards/meter/std": 0.16734978556632996, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8936691284179688, "rewards/total_composite/std": 0.16734978556632996, "reward": 0.8936691284179688, "reward_std": 0.16734978556632996, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03492492809891701, "sampling/sampling_logp_difference/max": 1.5639657974243164, "sampling/importance_sampling_ratio/min": 0.20930436253547668, "sampling/importance_sampling_ratio/mean": 1.0026055574417114, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1802924033254385, "clip_ratio/low_mean": 0.012298387009650469, "clip_ratio/low_min": 0.012298387009650469, "clip_ratio/high_mean": 0.021643388201482594, "clip_ratio/high_max": 0.021643388201482594, "clip_ratio/region_mean": 0.03394177521113306, "reward_total_mean": 0.8936691284179688, "reward_meter_mean": 0.8936691284179688, "reward_meter_std": 0.16734978556632996, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8936691284179688, "reward_total_composite_std": 0.16734978556632996} {"timestamp_utc": "2026-04-11T23:33:19Z", "mode": "train", "global_step": 1094, "epoch": 0.043941037072739685, "loss": 0.0321, "grad_norm": 2.7043662071228027, "learning_rate": 6.687878787878788e-06, "num_tokens": 2467653.0, "completions/mean_length": 136.0, "completions/min_length": 125.0, "completions/max_length": 139.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 136.0, "completions/min_terminated_length": 125.0, "completions/max_terminated_length": 139.0, "rewards/meter/mean": 0.7940855622291565, "rewards/meter/std": 0.11631763726472855, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6944444179534912, "rewards/repeat_penalty/std": 0.05143444612622261, "rewards/total_composite/mean": 0.41557225584983826, "rewards/total_composite/std": 0.08242405205965042, "reward": 0.41557225584983826, "reward_std": 0.08242405205965042, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01717483066022396, "sampling/sampling_logp_difference/max": 1.4296681880950928, "sampling/importance_sampling_ratio/min": 0.2393883466720581, "sampling/importance_sampling_ratio/mean": 0.9989469051361084, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07300170278176665, "clip_ratio/low_mean": 0.0017985611921176314, "clip_ratio/low_min": 0.0017985611921176314, "clip_ratio/high_mean": 0.011392419459298253, "clip_ratio/high_max": 0.011392419459298253, "clip_ratio/region_mean": 0.013190980651415884, "reward_total_mean": 0.41557225584983826, "reward_meter_mean": 0.7940855622291565, "reward_meter_std": 0.11631763726472855, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6944444179534912, "reward_repeat_penalty_std": 0.05143444612622261, "reward_total_composite_mean": 0.41557225584983826, "reward_total_composite_std": 0.08242405205965042} {"timestamp_utc": "2026-04-11T23:33:24Z", "mode": "train", "global_step": 1095, "epoch": 0.04398120255452464, "loss": -0.0163, "grad_norm": 11.521310806274414, "learning_rate": 6.684848484848485e-06, "num_tokens": 2469179.0, "completions/mean_length": 31.75, "completions/min_length": 30.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.75, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.8792948722839355, "rewards/meter/std": 0.25036633014678955, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8792948722839355, "rewards/total_composite/std": 0.25036633014678955, "reward": 0.8792948722839355, "reward_std": 0.25036633014678955, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04188957437872887, "sampling/sampling_logp_difference/max": 1.4607782363891602, "sampling/importance_sampling_ratio/min": 0.23205561935901642, "sampling/importance_sampling_ratio/mean": 0.9937551617622375, "sampling/importance_sampling_ratio/max": 1.482624888420105, "entropy": 0.16218565311282873, "clip_ratio/low_mean": 0.008333333767950535, "clip_ratio/low_min": 0.008333333767950535, "clip_ratio/high_mean": 0.031376007944345474, "clip_ratio/high_max": 0.031376007944345474, "clip_ratio/region_mean": 0.03970934171229601, "reward_total_mean": 0.8792948722839355, "reward_meter_mean": 0.8792948722839355, "reward_meter_std": 0.25036633014678955, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8792948722839355, "reward_total_composite_std": 0.25036633014678955} {"timestamp_utc": "2026-04-11T23:33:29Z", "mode": "train", "global_step": 1096, "epoch": 0.04402136803630959, "loss": 0.0349, "grad_norm": 4.185685634613037, "learning_rate": 6.681818181818183e-06, "num_tokens": 2471195.0, "completions/mean_length": 81.0, "completions/min_length": 74.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.0, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.8235281705856323, "rewards/meter/std": 0.1984855979681015, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.7183074951171875, "rewards/total_composite/std": 0.19789229333400726, "reward": 0.7183074951171875, "reward_std": 0.19789229333400726, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03307102620601654, "sampling/sampling_logp_difference/max": 1.3362131118774414, "sampling/importance_sampling_ratio/min": 0.262839138507843, "sampling/importance_sampling_ratio/mean": 0.9967372417449951, "sampling/importance_sampling_ratio/max": 1.7130012512207031, "entropy": 0.15196853037923574, "clip_ratio/low_mean": 0.005780933075584471, "clip_ratio/low_min": 0.005780933075584471, "clip_ratio/high_mean": 0.019367096945643425, "clip_ratio/high_max": 0.019367096945643425, "clip_ratio/region_mean": 0.025148030021227896, "reward_total_mean": 0.7183074951171875, "reward_meter_mean": 0.8235281705856323, "reward_meter_std": 0.1984855979681015, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.7183074951171875, "reward_total_composite_std": 0.19789229333400726} {"timestamp_utc": "2026-04-11T23:33:37Z", "mode": "train", "global_step": 1097, "epoch": 0.044061533518094546, "loss": -0.0005, "grad_norm": 1.8824577331542969, "learning_rate": 6.678787878787879e-06, "num_tokens": 2475359.0, "completions/mean_length": 304.5, "completions/min_length": 280.0, "completions/max_length": 320.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 304.5, "completions/min_terminated_length": 280.0, "completions/max_terminated_length": 320.0, "rewards/meter/mean": 0.953475832939148, "rewards/meter/std": 0.11951109021902084, "rewards/count_adherence/mean": 0.7403846383094788, "rewards/count_adherence/std": 0.039811473339796066, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6966374516487122, "rewards/repeat_penalty/std": 0.09851117432117462, "rewards/total_composite/mean": 0.49607303738594055, "rewards/total_composite/std": 0.11844708025455475, "reward": 0.49607303738594055, "reward_std": 0.11844708025455475, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027330797165632248, "sampling/sampling_logp_difference/max": 2.8147997856140137, "sampling/importance_sampling_ratio/min": 0.05991671606898308, "sampling/importance_sampling_ratio/mean": 1.0036851167678833, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.16177368070930243, "clip_ratio/low_mean": 0.010694848489947617, "clip_ratio/low_min": 0.010694848489947617, "clip_ratio/high_mean": 0.01060027233324945, "clip_ratio/high_max": 0.01060027233324945, "clip_ratio/region_mean": 0.021295120823197067, "reward_total_mean": 0.49607303738594055, "reward_meter_mean": 0.953475832939148, "reward_meter_std": 0.11951109021902084, "reward_count_adherence_mean": 0.7403846383094788, "reward_count_adherence_std": 0.039811473339796066, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6966374516487122, "reward_repeat_penalty_std": 0.09851117432117462, "reward_total_composite_mean": 0.49607303738594055, "reward_total_composite_std": 0.11844708025455475} {"timestamp_utc": "2026-04-11T23:33:41Z", "mode": "train", "global_step": 1098, "epoch": 0.0441016989998795, "loss": 0.0173, "grad_norm": 6.93865966796875, "learning_rate": 6.6757575757575766e-06, "num_tokens": 2477124.0, "completions/mean_length": 61.625, "completions/min_length": 60.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.625, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.876738965511322, "rewards/meter/std": 0.2505471706390381, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.876738965511322, "rewards/total_composite/std": 0.2505471706390381, "reward": 0.876738965511322, "reward_std": 0.2505471408367157, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026354925706982613, "sampling/sampling_logp_difference/max": 1.0480327606201172, "sampling/importance_sampling_ratio/min": 0.3506268262863159, "sampling/importance_sampling_ratio/mean": 1.0015573501586914, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08917439868673682, "clip_ratio/low_mean": 0.0019841270986944437, "clip_ratio/low_min": 0.0019841270986944437, "clip_ratio/high_mean": 0.014381667831912637, "clip_ratio/high_max": 0.014381667831912637, "clip_ratio/region_mean": 0.01636579493060708, "reward_total_mean": 0.876738965511322, "reward_meter_mean": 0.876738965511322, "reward_meter_std": 0.2505471706390381, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.876738965511322, "reward_total_composite_std": 0.2505471706390381} {"timestamp_utc": "2026-04-11T23:33:48Z", "mode": "train", "global_step": 1099, "epoch": 0.044141864481664454, "loss": -0.0507, "grad_norm": 3.1013591289520264, "learning_rate": 6.672727272727273e-06, "num_tokens": 2480198.0, "completions/mean_length": 205.25, "completions/min_length": 171.0, "completions/max_length": 224.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 205.25, "completions/min_terminated_length": 171.0, "completions/max_terminated_length": 224.0, "rewards/meter/mean": 0.7326644062995911, "rewards/meter/std": 0.39641380310058594, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6022727489471436, "rewards/repeat_penalty/std": 0.14527180790901184, "rewards/total_composite/mean": 0.373015820980072, "rewards/total_composite/std": 0.22151786088943481, "reward": 0.373015820980072, "reward_std": 0.22151786088943481, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017836585640907288, "sampling/sampling_logp_difference/max": 3.5347535610198975, "sampling/importance_sampling_ratio/min": 0.02916594408452511, "sampling/importance_sampling_ratio/mean": 1.0023084878921509, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09231163933873177, "clip_ratio/low_mean": 0.0041160593973472714, "clip_ratio/low_min": 0.0041160593973472714, "clip_ratio/high_mean": 0.009313070855569094, "clip_ratio/high_max": 0.009313070855569094, "clip_ratio/region_mean": 0.013429130252916366, "reward_total_mean": 0.373015820980072, "reward_meter_mean": 0.7326644062995911, "reward_meter_std": 0.39641380310058594, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6022727489471436, "reward_repeat_penalty_std": 0.14527180790901184, "reward_total_composite_mean": 0.373015820980072, "reward_total_composite_std": 0.22151786088943481} {"timestamp_utc": "2026-04-11T23:33:55Z", "mode": "train", "global_step": 1100, "epoch": 0.04418202996344941, "loss": -0.0248, "grad_norm": 0.9629241228103638, "learning_rate": 6.66969696969697e-06, "num_tokens": 2484623.0, "completions/mean_length": 327.125, "completions/min_length": 302.0, "completions/max_length": 345.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 327.125, "completions/min_terminated_length": 302.0, "completions/max_terminated_length": 345.0, "rewards/meter/mean": 0.993415355682373, "rewards/meter/std": 0.007653866894543171, "rewards/count_adherence/mean": 0.7291666865348816, "rewards/count_adherence/std": 0.038575831800699234, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5248161554336548, "rewards/repeat_penalty/std": 0.21153290569782257, "rewards/total_composite/mean": 0.386793315410614, "rewards/total_composite/std": 0.16382494568824768, "reward": 0.386793315410614, "reward_std": 0.16382494568824768, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010836898349225521, "sampling/sampling_logp_difference/max": 1.045947790145874, "sampling/importance_sampling_ratio/min": 0.35135865211486816, "sampling/importance_sampling_ratio/mean": 1.001177430152893, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06283333199098706, "clip_ratio/low_mean": 0.001618139911442995, "clip_ratio/low_min": 0.001618139911442995, "clip_ratio/high_mean": 0.010039791552117094, "clip_ratio/high_max": 0.010039791552117094, "clip_ratio/region_mean": 0.01165793146356009, "reward_total_mean": 0.386793315410614, "reward_meter_mean": 0.993415355682373, "reward_meter_std": 0.007653866894543171, "reward_count_adherence_mean": 0.7291666865348816, "reward_count_adherence_std": 0.038575831800699234, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5248161554336548, "reward_repeat_penalty_std": 0.21153290569782257, "reward_total_composite_mean": 0.386793315410614, "reward_total_composite_std": 0.16382494568824768} {"timestamp_utc": "2026-04-11T23:34:59Z", "mode": "eval", "global_step": 1100, "epoch": 0.04418202996344941, "eval_loss": NaN, "eval_runtime": 63.6923, "eval_samples_per_second": 1.633, "eval_steps_per_second": 0.204, "eval_num_tokens": 2484623.0, "eval_completions/mean_length": 188.08653846153845, "eval_completions/min_length": 57.92307692307692, "eval_completions/max_length": 328.61538461538464, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 188.08653846153845, "eval_completions/min_terminated_length": 57.92307692307692, "eval_completions/max_terminated_length": 328.61538461538464, "eval_rewards/meter/mean": 0.6326167858563937, "eval_rewards/meter/std": 0.35957076343206257, "eval_rewards/count_adherence/mean": 0.8649956675676199, "eval_rewards/count_adherence/std": 0.12389520039925209, "eval_rewards/arabic_clean/mean": 0.9807692307692307, "eval_rewards/arabic_clean/std": 0.05439282839114849, "eval_rewards/repeat_penalty/mean": 0.6625011425751907, "eval_rewards/repeat_penalty/std": 0.24176405255611128, "eval_rewards/total_composite/mean": 0.385673477099492, "eval_rewards/total_composite/std": 0.3103876068041875, "eval_reward": 0.385673477099492, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.008682384119870571, "eval_sampling/sampling_logp_difference/max": 0.9079764164411105, "eval_sampling/importance_sampling_ratio/min": 0.4163260803772853, "eval_sampling/importance_sampling_ratio/mean": 1.0015292717860296, "eval_sampling/importance_sampling_ratio/max": 1.3779725111447847, "eval_entropy": 0.07903132902888152, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.385673477099492, "eval_reward_meter_mean": 0.6326167858563937, "eval_reward_meter_std": 0.35957076343206257, "eval_reward_count_adherence_mean": 0.8649956675676199, "eval_reward_count_adherence_std": 0.12389520039925209, "eval_reward_arabic_clean_mean": 0.9807692307692307, "eval_reward_arabic_clean_std": 0.05439282839114849, "eval_reward_repeat_penalty_mean": 0.6625011425751907, "eval_reward_repeat_penalty_std": 0.24176405255611128, "eval_reward_total_composite_mean": 0.385673477099492, "eval_reward_total_composite_std": 0.3103876068041875} {"timestamp_utc": "2026-04-11T23:35:07Z", "mode": "train", "global_step": 1101, "epoch": 0.04422219544523436, "loss": 0.0008, "grad_norm": 7.680727958679199, "learning_rate": 6.666666666666667e-06, "num_tokens": 2486422.0, "completions/mean_length": 55.875, "completions/min_length": 50.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.875, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.8475104570388794, "rewards/meter/std": 0.17176933586597443, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8475104570388794, "rewards/total_composite/std": 0.17176933586597443, "reward": 0.8475104570388794, "reward_std": 0.17176933586597443, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05723501741886139, "sampling/sampling_logp_difference/max": 2.4088823795318604, "sampling/importance_sampling_ratio/min": 0.08991573750972748, "sampling/importance_sampling_ratio/mean": 1.0003966093063354, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.22917167469859123, "clip_ratio/low_mean": 0.01142857177183032, "clip_ratio/low_min": 0.01142857177183032, "clip_ratio/high_mean": 0.02208862197585404, "clip_ratio/high_max": 0.02208862197585404, "clip_ratio/region_mean": 0.03351719374768436, "reward_total_mean": 0.8475104570388794, "reward_meter_mean": 0.8475104570388794, "reward_meter_std": 0.17176933586597443, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8475104570388794, "reward_total_composite_std": 0.17176933586597443} {"timestamp_utc": "2026-04-11T23:35:15Z", "mode": "train", "global_step": 1102, "epoch": 0.04426236092701932, "loss": -0.0479, "grad_norm": 9.668932914733887, "learning_rate": 6.663636363636365e-06, "num_tokens": 2490583.0, "completions/mean_length": 296.125, "completions/min_length": 244.0, "completions/max_length": 335.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 296.125, "completions/min_terminated_length": 244.0, "completions/max_terminated_length": 335.0, "rewards/meter/mean": 0.7127225399017334, "rewards/meter/std": 0.37384799122810364, "rewards/count_adherence/mean": 0.9722222089767456, "rewards/count_adherence/std": 0.05143444985151291, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5183823704719543, "rewards/repeat_penalty/std": 0.17379139363765717, "rewards/total_composite/mean": 0.37116703391075134, "rewards/total_composite/std": 0.24903467297554016, "reward": 0.37116703391075134, "reward_std": 0.24903464317321777, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014599828980863094, "sampling/sampling_logp_difference/max": 4.119277000427246, "sampling/importance_sampling_ratio/min": 0.016256263479590416, "sampling/importance_sampling_ratio/mean": 1.0006206035614014, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.054934965912252665, "clip_ratio/low_mean": 0.004965955973602831, "clip_ratio/low_min": 0.004965955973602831, "clip_ratio/high_mean": 0.007919449737528339, "clip_ratio/high_max": 0.007919449737528339, "clip_ratio/region_mean": 0.01288540571113117, "reward_total_mean": 0.37116703391075134, "reward_meter_mean": 0.7127225399017334, "reward_meter_std": 0.37384799122810364, "reward_count_adherence_mean": 0.9722222089767456, "reward_count_adherence_std": 0.05143444985151291, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5183823704719543, "reward_repeat_penalty_std": 0.17379139363765717, "reward_total_composite_mean": 0.37116703391075134, "reward_total_composite_std": 0.24903467297554016} {"timestamp_utc": "2026-04-11T23:35:20Z", "mode": "train", "global_step": 1103, "epoch": 0.04430252640880428, "loss": -0.0174, "grad_norm": 3.663283586502075, "learning_rate": 6.660606060606061e-06, "num_tokens": 2492366.0, "completions/mean_length": 66.875, "completions/min_length": 64.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.875, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.49698805809020996, "rewards/meter/std": 0.368730753660202, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.49698805809020996, "rewards/total_composite/std": 0.368730753660202, "reward": 0.49698805809020996, "reward_std": 0.3687307834625244, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02970782294869423, "sampling/sampling_logp_difference/max": 1.8465352058410645, "sampling/importance_sampling_ratio/min": 0.25461694598197937, "sampling/importance_sampling_ratio/mean": 0.9964687824249268, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10305365733802319, "clip_ratio/low_mean": 0.005657031899318099, "clip_ratio/low_min": 0.005657031899318099, "clip_ratio/high_mean": 0.012646116432733834, "clip_ratio/high_max": 0.012646116432733834, "clip_ratio/region_mean": 0.018303148332051933, "reward_total_mean": 0.49698805809020996, "reward_meter_mean": 0.49698805809020996, "reward_meter_std": 0.368730753660202, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.49698805809020996, "reward_total_composite_std": 0.368730753660202} {"timestamp_utc": "2026-04-11T23:35:25Z", "mode": "train", "global_step": 1104, "epoch": 0.04434269189058923, "loss": -0.0068, "grad_norm": 7.544509410858154, "learning_rate": 6.657575757575758e-06, "num_tokens": 2494079.0, "completions/mean_length": 51.125, "completions/min_length": 47.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.125, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.6021931171417236, "rewards/meter/std": 0.43061062693595886, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6021931171417236, "rewards/total_composite/std": 0.43061062693595886, "reward": 0.6021931171417236, "reward_std": 0.4306105971336365, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05377590283751488, "sampling/sampling_logp_difference/max": 1.3904236555099487, "sampling/importance_sampling_ratio/min": 0.24896980822086334, "sampling/importance_sampling_ratio/mean": 0.9914664030075073, "sampling/importance_sampling_ratio/max": 1.6187946796417236, "entropy": 0.18264612928032875, "clip_ratio/low_mean": 0.01027012919075787, "clip_ratio/low_min": 0.01027012919075787, "clip_ratio/high_mean": 0.028512453194707632, "clip_ratio/high_max": 0.028512453194707632, "clip_ratio/region_mean": 0.0387825823854655, "reward_total_mean": 0.6021931171417236, "reward_meter_mean": 0.6021931171417236, "reward_meter_std": 0.43061062693595886, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6021931171417236, "reward_total_composite_std": 0.43061062693595886} {"timestamp_utc": "2026-04-11T23:35:30Z", "mode": "train", "global_step": 1105, "epoch": 0.044382857372374185, "loss": 0.0134, "grad_norm": 8.104043960571289, "learning_rate": 6.654545454545455e-06, "num_tokens": 2495813.0, "completions/mean_length": 51.75, "completions/min_length": 48.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.75, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.8142023086547852, "rewards/meter/std": 0.21253840625286102, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8142023086547852, "rewards/total_composite/std": 0.21253840625286102, "reward": 0.8142023086547852, "reward_std": 0.21253842115402222, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06759097427129745, "sampling/sampling_logp_difference/max": 1.6661019325256348, "sampling/importance_sampling_ratio/min": 0.188982293009758, "sampling/importance_sampling_ratio/mean": 1.0077879428863525, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2174948314204812, "clip_ratio/low_mean": 0.00491898157633841, "clip_ratio/low_min": 0.00491898157633841, "clip_ratio/high_mean": 0.035976887214928865, "clip_ratio/high_max": 0.035976887214928865, "clip_ratio/region_mean": 0.040895868791267276, "reward_total_mean": 0.8142023086547852, "reward_meter_mean": 0.8142023086547852, "reward_meter_std": 0.21253840625286102, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8142023086547852, "reward_total_composite_std": 0.21253840625286102} {"timestamp_utc": "2026-04-11T23:35:38Z", "mode": "train", "global_step": 1106, "epoch": 0.04442302285415914, "loss": -0.0234, "grad_norm": 1.9001294374465942, "learning_rate": 6.651515151515152e-06, "num_tokens": 2499568.0, "completions/mean_length": 275.375, "completions/min_length": 256.0, "completions/max_length": 301.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 275.375, "completions/min_terminated_length": 256.0, "completions/max_terminated_length": 301.0, "rewards/meter/mean": 0.8482787013053894, "rewards/meter/std": 0.33679211139678955, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.05050762742757797, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6220238208770752, "rewards/repeat_penalty/std": 0.03127124905586243, "rewards/total_composite/mean": 0.46502256393432617, "rewards/total_composite/std": 0.1884276121854782, "reward": 0.46502256393432617, "reward_std": 0.1884276121854782, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012011334300041199, "sampling/sampling_logp_difference/max": 1.7623469829559326, "sampling/importance_sampling_ratio/min": 0.171641543507576, "sampling/importance_sampling_ratio/mean": 1.0009040832519531, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.05004570330493152, "clip_ratio/low_mean": 0.00146484375, "clip_ratio/low_min": 0.00146484375, "clip_ratio/high_mean": 0.00831845449283719, "clip_ratio/high_max": 0.00831845449283719, "clip_ratio/region_mean": 0.00978329824283719, "reward_total_mean": 0.46502256393432617, "reward_meter_mean": 0.8482787013053894, "reward_meter_std": 0.33679211139678955, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.05050762742757797, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6220238208770752, "reward_repeat_penalty_std": 0.03127124905586243, "reward_total_composite_mean": 0.46502256393432617, "reward_total_composite_std": 0.1884276121854782} {"timestamp_utc": "2026-04-11T23:35:42Z", "mode": "train", "global_step": 1107, "epoch": 0.04446318833594409, "loss": 0.0015, "grad_norm": 2.1617753505706787, "learning_rate": 6.6484848484848485e-06, "num_tokens": 2501018.0, "completions/mean_length": 38.25, "completions/min_length": 38.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.25, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.92488694190979, "rewards/meter/std": 0.09670114517211914, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.92488694190979, "rewards/total_composite/std": 0.09670114517211914, "reward": 0.92488694190979, "reward_std": 0.09670114517211914, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012348960153758526, "sampling/sampling_logp_difference/max": 0.6579999923706055, "sampling/importance_sampling_ratio/min": 0.5178860425949097, "sampling/importance_sampling_ratio/mean": 1.0063523054122925, "sampling/importance_sampling_ratio/max": 1.7915105819702148, "entropy": 0.07856028527021408, "clip_ratio/low_mean": 0.003289473708719015, "clip_ratio/low_min": 0.003289473708719015, "clip_ratio/high_mean": 0.009699730202555656, "clip_ratio/high_max": 0.009699730202555656, "clip_ratio/region_mean": 0.012989203911274672, "reward_total_mean": 0.92488694190979, "reward_meter_mean": 0.92488694190979, "reward_meter_std": 0.09670114517211914, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.92488694190979, "reward_total_composite_std": 0.09670114517211914} {"timestamp_utc": "2026-04-11T23:35:47Z", "mode": "train", "global_step": 1108, "epoch": 0.04450335381772905, "loss": -0.0307, "grad_norm": 6.412082672119141, "learning_rate": 6.645454545454546e-06, "num_tokens": 2503004.0, "completions/mean_length": 79.25, "completions/min_length": 72.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.25, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.4985108971595764, "rewards/meter/std": 0.402507483959198, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.17728103697299957, "rewards/total_composite/mean": 0.39024245738983154, "rewards/total_composite/std": 0.367125540971756, "reward": 0.39024245738983154, "reward_std": 0.3671255111694336, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.043393637984991074, "sampling/sampling_logp_difference/max": 2.8125345706939697, "sampling/importance_sampling_ratio/min": 0.06005259230732918, "sampling/importance_sampling_ratio/mean": 1.0030810832977295, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2001291485503316, "clip_ratio/low_mean": 0.014691225020214915, "clip_ratio/low_min": 0.014691225020214915, "clip_ratio/high_mean": 0.015104167046956718, "clip_ratio/high_max": 0.015104167046956718, "clip_ratio/region_mean": 0.029795392067171633, "reward_total_mean": 0.39024245738983154, "reward_meter_mean": 0.4985108971595764, "reward_meter_std": 0.402507483959198, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.17728103697299957, "reward_total_composite_mean": 0.39024245738983154, "reward_total_composite_std": 0.367125540971756} {"timestamp_utc": "2026-04-11T23:35:51Z", "mode": "train", "global_step": 1109, "epoch": 0.044543519299514, "loss": -0.025, "grad_norm": 8.44660472869873, "learning_rate": 6.642424242424242e-06, "num_tokens": 2504551.0, "completions/mean_length": 47.375, "completions/min_length": 42.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.375, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.9353389143943787, "rewards/meter/std": 0.06214950606226921, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9353389143943787, "rewards/total_composite/std": 0.06214950606226921, "reward": 0.9353389143943787, "reward_std": 0.062149498611688614, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.043186187744140625, "sampling/sampling_logp_difference/max": 1.5208110809326172, "sampling/importance_sampling_ratio/min": 0.21853457391262054, "sampling/importance_sampling_ratio/mean": 0.9965187907218933, "sampling/importance_sampling_ratio/max": 1.534562110900879, "entropy": 0.19667030405253172, "clip_ratio/low_mean": 0.01064311619848013, "clip_ratio/low_min": 0.01064311619848013, "clip_ratio/high_mean": 0.02806122461333871, "clip_ratio/high_max": 0.02806122461333871, "clip_ratio/region_mean": 0.03870434081181884, "reward_total_mean": 0.9353389143943787, "reward_meter_mean": 0.9353389143943787, "reward_meter_std": 0.06214950606226921, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9353389143943787, "reward_total_composite_std": 0.06214950606226921} {"timestamp_utc": "2026-04-11T23:35:57Z", "mode": "train", "global_step": 1110, "epoch": 0.044583684781298955, "loss": -0.022, "grad_norm": 4.039949893951416, "learning_rate": 6.63939393939394e-06, "num_tokens": 2507175.0, "completions/mean_length": 150.0, "completions/min_length": 139.0, "completions/max_length": 160.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 150.0, "completions/min_terminated_length": 139.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.9286544322967529, "rewards/meter/std": 0.16951557993888855, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8333333730697632, "rewards/repeat_penalty/std": 0.059391383081674576, "rewards/total_composite/mean": 0.5778908729553223, "rewards/total_composite/std": 0.1044817790389061, "reward": 0.5778908729553223, "reward_std": 0.1044817790389061, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.038920193910598755, "sampling/sampling_logp_difference/max": 1.9175208806991577, "sampling/importance_sampling_ratio/min": 0.14697086811065674, "sampling/importance_sampling_ratio/mean": 0.9992915987968445, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17426198534667492, "clip_ratio/low_mean": 0.009535853168927133, "clip_ratio/low_min": 0.009535853168927133, "clip_ratio/high_mean": 0.024372577434405684, "clip_ratio/high_max": 0.024372577434405684, "clip_ratio/region_mean": 0.03390843060333282, "reward_total_mean": 0.5778908729553223, "reward_meter_mean": 0.9286544322967529, "reward_meter_std": 0.16951557993888855, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8333333730697632, "reward_repeat_penalty_std": 0.059391383081674576, "reward_total_composite_mean": 0.5778908729553223, "reward_total_composite_std": 0.1044817790389061} {"timestamp_utc": "2026-04-11T23:36:01Z", "mode": "train", "global_step": 1111, "epoch": 0.04462385026308391, "loss": 0.0099, "grad_norm": 3.789764881134033, "learning_rate": 6.6363636363636375e-06, "num_tokens": 2508931.0, "completions/mean_length": 73.5, "completions/min_length": 73.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.5, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9711061716079712, "rewards/meter/std": 0.02328452095389366, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9711061716079712, "rewards/total_composite/std": 0.02328452095389366, "reward": 0.9711061716079712, "reward_std": 0.023284511640667915, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011047722771763802, "sampling/sampling_logp_difference/max": 1.3545985221862793, "sampling/importance_sampling_ratio/min": 0.2580508887767792, "sampling/importance_sampling_ratio/mean": 1.0022356510162354, "sampling/importance_sampling_ratio/max": 1.7336740493774414, "entropy": 0.03971482953056693, "clip_ratio/low_mean": 0.0050675676902756095, "clip_ratio/low_min": 0.0050675676902756095, "clip_ratio/high_mean": 0.0051369862630963326, "clip_ratio/high_max": 0.0051369862630963326, "clip_ratio/region_mean": 0.010204553953371942, "reward_total_mean": 0.9711061716079712, "reward_meter_mean": 0.9711061716079712, "reward_meter_std": 0.02328452095389366, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9711061716079712, "reward_total_composite_std": 0.02328452095389366} {"timestamp_utc": "2026-04-11T23:36:06Z", "mode": "train", "global_step": 1112, "epoch": 0.04466401574486886, "loss": -0.0457, "grad_norm": 0.8946425914764404, "learning_rate": 6.633333333333334e-06, "num_tokens": 2511051.0, "completions/mean_length": 106.0, "completions/min_length": 92.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.0, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.9727685451507568, "rewards/meter/std": 0.053206540644168854, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7782148122787476, "rewards/total_composite/std": 0.04256521537899971, "reward": 0.7782148122787476, "reward_std": 0.04256521537899971, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0040269712917506695, "sampling/sampling_logp_difference/max": 0.56025230884552, "sampling/importance_sampling_ratio/min": 0.5710649490356445, "sampling/importance_sampling_ratio/mean": 0.9997245073318481, "sampling/importance_sampling_ratio/max": 1.4477814435958862, "entropy": 0.01866412186063826, "clip_ratio/low_mean": 0.0027173913549631834, "clip_ratio/low_min": 0.0027173913549631834, "clip_ratio/high_mean": 0.0034722222480922937, "clip_ratio/high_max": 0.0034722222480922937, "clip_ratio/region_mean": 0.006189613603055477, "reward_total_mean": 0.7782148122787476, "reward_meter_mean": 0.9727685451507568, "reward_meter_std": 0.053206540644168854, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7782148122787476, "reward_total_composite_std": 0.04256521537899971} {"timestamp_utc": "2026-04-11T23:36:10Z", "mode": "train", "global_step": 1113, "epoch": 0.044704181226653816, "loss": -0.0282, "grad_norm": 6.619121074676514, "learning_rate": 6.630303030303031e-06, "num_tokens": 2512391.0, "completions/mean_length": 23.5, "completions/min_length": 23.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.5, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.867714524269104, "rewards/meter/std": 0.052471715956926346, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.867714524269104, "rewards/total_composite/std": 0.052471715956926346, "reward": 0.867714524269104, "reward_std": 0.052471719682216644, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024627521634101868, "sampling/sampling_logp_difference/max": 0.767571210861206, "sampling/importance_sampling_ratio/min": 0.4641389846801758, "sampling/importance_sampling_ratio/mean": 1.008794903755188, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08848860999569297, "clip_ratio/low_mean": 0.010869565419852734, "clip_ratio/low_min": 0.010869565419852734, "clip_ratio/high_mean": 0.010064412374049425, "clip_ratio/high_max": 0.010064412374049425, "clip_ratio/region_mean": 0.02093397779390216, "reward_total_mean": 0.867714524269104, "reward_meter_mean": 0.867714524269104, "reward_meter_std": 0.052471715956926346, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.867714524269104, "reward_total_composite_std": 0.052471715956926346} {"timestamp_utc": "2026-04-11T23:36:15Z", "mode": "train", "global_step": 1114, "epoch": 0.04474434670843877, "loss": -0.0256, "grad_norm": 11.225849151611328, "learning_rate": 6.627272727272728e-06, "num_tokens": 2513999.0, "completions/mean_length": 43.0, "completions/min_length": 41.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.0, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.8079327940940857, "rewards/meter/std": 0.1427396684885025, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8079327940940857, "rewards/total_composite/std": 0.1427396684885025, "reward": 0.8079327940940857, "reward_std": 0.1427396684885025, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04281027242541313, "sampling/sampling_logp_difference/max": 0.9714531898498535, "sampling/importance_sampling_ratio/min": 0.3785325586795807, "sampling/importance_sampling_ratio/mean": 1.0110169649124146, "sampling/importance_sampling_ratio/max": 1.9035590887069702, "entropy": 0.15364955179393291, "clip_ratio/low_mean": 0.014953542733564973, "clip_ratio/low_min": 0.014953542733564973, "clip_ratio/high_mean": 0.017787929391488433, "clip_ratio/high_max": 0.017787929391488433, "clip_ratio/region_mean": 0.032741472125053406, "reward_total_mean": 0.8079327940940857, "reward_meter_mean": 0.8079327940940857, "reward_meter_std": 0.1427396684885025, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8079327940940857, "reward_total_composite_std": 0.1427396684885025} {"timestamp_utc": "2026-04-11T23:36:19Z", "mode": "train", "global_step": 1115, "epoch": 0.044784512190223724, "loss": -0.0173, "grad_norm": 7.548732757568359, "learning_rate": 6.624242424242425e-06, "num_tokens": 2515720.0, "completions/mean_length": 49.125, "completions/min_length": 44.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.125, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.7826473712921143, "rewards/meter/std": 0.3253748416900635, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7826473712921143, "rewards/total_composite/std": 0.3253748416900635, "reward": 0.7826473712921143, "reward_std": 0.3253748118877411, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.051869362592697144, "sampling/sampling_logp_difference/max": 1.9477629661560059, "sampling/importance_sampling_ratio/min": 0.14259269833564758, "sampling/importance_sampling_ratio/mean": 1.001320719718933, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1894742762669921, "clip_ratio/low_mean": 0.00828598509542644, "clip_ratio/low_min": 0.00828598509542644, "clip_ratio/high_mean": 0.029880503891035914, "clip_ratio/high_max": 0.029880503891035914, "clip_ratio/region_mean": 0.038166488986462355, "reward_total_mean": 0.7826473712921143, "reward_meter_mean": 0.7826473712921143, "reward_meter_std": 0.3253748416900635, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7826473712921143, "reward_total_composite_std": 0.3253748416900635} {"timestamp_utc": "2026-04-11T23:36:24Z", "mode": "train", "global_step": 1116, "epoch": 0.04482467767200868, "loss": 0.0026, "grad_norm": 8.76555347442627, "learning_rate": 6.621212121212121e-06, "num_tokens": 2517481.0, "completions/mean_length": 73.125, "completions/min_length": 73.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.125, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9871143698692322, "rewards/meter/std": 0.010886327363550663, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9871143698692322, "rewards/total_composite/std": 0.010886327363550663, "reward": 0.9871143698692322, "reward_std": 0.010886335745453835, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011934218928217888, "sampling/sampling_logp_difference/max": 1.53245210647583, "sampling/importance_sampling_ratio/min": 0.2160053551197052, "sampling/importance_sampling_ratio/mean": 0.997620165348053, "sampling/importance_sampling_ratio/max": 1.276580810546875, "entropy": 0.03595073730684817, "clip_ratio/low_mean": 0.0017123287543654442, "clip_ratio/low_min": 0.0017123287543654442, "clip_ratio/high_mean": 0.0034246575087308884, "clip_ratio/high_max": 0.0034246575087308884, "clip_ratio/region_mean": 0.0051369862630963326, "reward_total_mean": 0.9871143698692322, "reward_meter_mean": 0.9871143698692322, "reward_meter_std": 0.010886327363550663, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9871143698692322, "reward_total_composite_std": 0.010886327363550663} {"timestamp_utc": "2026-04-11T23:36:32Z", "mode": "train", "global_step": 1117, "epoch": 0.04486484315379363, "loss": 0.0033, "grad_norm": 0.39119431376457214, "learning_rate": 6.618181818181819e-06, "num_tokens": 2521793.0, "completions/mean_length": 341.0, "completions/min_length": 335.0, "completions/max_length": 360.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 341.0, "completions/min_terminated_length": 335.0, "completions/max_terminated_length": 360.0, "rewards/meter/mean": 0.9725086688995361, "rewards/meter/std": 0.01511977519840002, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5882353186607361, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.42904794216156006, "rewards/total_composite/std": 0.006670483388006687, "reward": 0.42904794216156006, "reward_std": 0.006670480594038963, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00816288124769926, "sampling/sampling_logp_difference/max": 2.6277527809143066, "sampling/importance_sampling_ratio/min": 0.07224062830209732, "sampling/importance_sampling_ratio/mean": 1.0016568899154663, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03286444069817662, "clip_ratio/low_mean": 0.001091206620912999, "clip_ratio/low_min": 0.001091206620912999, "clip_ratio/high_mean": 0.002917782054282725, "clip_ratio/high_max": 0.002917782054282725, "clip_ratio/region_mean": 0.004008988675195724, "reward_total_mean": 0.42904794216156006, "reward_meter_mean": 0.9725086688995361, "reward_meter_std": 0.01511977519840002, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5882353186607361, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.42904794216156006, "reward_total_composite_std": 0.006670483388006687} {"timestamp_utc": "2026-04-11T23:36:38Z", "mode": "train", "global_step": 1118, "epoch": 0.044905008635578586, "loss": -0.0082, "grad_norm": 4.05780553817749, "learning_rate": 6.615151515151516e-06, "num_tokens": 2523775.0, "completions/mean_length": 86.75, "completions/min_length": 82.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.75, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.7943093776702881, "rewards/meter/std": 0.21005631983280182, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.6824886798858643, "rewards/total_composite/std": 0.2206040620803833, "reward": 0.6824886798858643, "reward_std": 0.2206040620803833, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028331898152828217, "sampling/sampling_logp_difference/max": 1.0872173309326172, "sampling/importance_sampling_ratio/min": 0.3371534049510956, "sampling/importance_sampling_ratio/mean": 0.9958962202072144, "sampling/importance_sampling_ratio/max": 1.9720497131347656, "entropy": 0.11351043824106455, "clip_ratio/low_mean": 0.0043270515743643045, "clip_ratio/low_min": 0.0043270515743643045, "clip_ratio/high_mean": 0.020123523310758173, "clip_ratio/high_max": 0.020123523310758173, "clip_ratio/region_mean": 0.024450574885122478, "reward_total_mean": 0.6824886798858643, "reward_meter_mean": 0.7943093776702881, "reward_meter_std": 0.21005631983280182, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.6824886798858643, "reward_total_composite_std": 0.2206040620803833} {"timestamp_utc": "2026-04-11T23:36:42Z", "mode": "train", "global_step": 1119, "epoch": 0.04494517411736354, "loss": 0.0076, "grad_norm": 4.082438945770264, "learning_rate": 6.612121212121213e-06, "num_tokens": 2525659.0, "completions/mean_length": 60.5, "completions/min_length": 60.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.5, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9949465990066528, "rewards/meter/std": 0.0008418544312007725, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9949465990066528, "rewards/total_composite/std": 0.0008418544312007725, "reward": 0.9949465990066528, "reward_std": 0.0008418540237471461, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016099898144602776, "sampling/sampling_logp_difference/max": 1.0732321739196777, "sampling/importance_sampling_ratio/min": 0.4875824749469757, "sampling/importance_sampling_ratio/mean": 1.0043621063232422, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.061238054651767015, "clip_ratio/low_mean": 0.012297285255044699, "clip_ratio/low_min": 0.012297285255044699, "clip_ratio/high_mean": 0.006250000325962901, "clip_ratio/high_max": 0.006250000325962901, "clip_ratio/region_mean": 0.0185472855810076, "reward_total_mean": 0.9949465990066528, "reward_meter_mean": 0.9949465990066528, "reward_meter_std": 0.0008418544312007725, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9949465990066528, "reward_total_composite_std": 0.0008418544312007725} {"timestamp_utc": "2026-04-11T23:36:47Z", "mode": "train", "global_step": 1120, "epoch": 0.044985339599148494, "loss": -0.0067, "grad_norm": 3.3688883781433105, "learning_rate": 6.609090909090909e-06, "num_tokens": 2527370.0, "completions/mean_length": 60.875, "completions/min_length": 60.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.875, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9956821203231812, "rewards/meter/std": 0.0002610879309941083, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9956821203231812, "rewards/total_composite/std": 0.0002610879309941083, "reward": 0.9956821203231812, "reward_std": 0.0002611094678286463, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012183277867734432, "sampling/sampling_logp_difference/max": 1.1166975498199463, "sampling/importance_sampling_ratio/min": 0.3273591101169586, "sampling/importance_sampling_ratio/mean": 1.0035232305526733, "sampling/importance_sampling_ratio/max": 1.5681954622268677, "entropy": 0.04563273023813963, "clip_ratio/low_mean": 0.0020833334419876337, "clip_ratio/low_min": 0.0020833334419876337, "clip_ratio/high_mean": 0.002016128972172737, "clip_ratio/high_max": 0.002016128972172737, "clip_ratio/region_mean": 0.004099462414160371, "reward_total_mean": 0.9956821203231812, "reward_meter_mean": 0.9956821203231812, "reward_meter_std": 0.0002610879309941083, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9956821203231812, "reward_total_composite_std": 0.0002610879309941083} {"timestamp_utc": "2026-04-11T23:36:51Z", "mode": "train", "global_step": 1121, "epoch": 0.04502550508093345, "loss": -0.0073, "grad_norm": 6.166923999786377, "learning_rate": 6.606060606060607e-06, "num_tokens": 2529160.0, "completions/mean_length": 69.75, "completions/min_length": 68.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.75, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.7864480018615723, "rewards/meter/std": 0.26775890588760376, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5, "rewards/repeat_penalty/std": 0.1511857956647873, "rewards/total_composite/mean": 0.3661602735519409, "rewards/total_composite/std": 0.13910025358200073, "reward": 0.3661602735519409, "reward_std": 0.13910023868083954, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01315419189631939, "sampling/sampling_logp_difference/max": 1.5172581672668457, "sampling/importance_sampling_ratio/min": 0.219312384724617, "sampling/importance_sampling_ratio/mean": 1.0017613172531128, "sampling/importance_sampling_ratio/max": 1.638248085975647, "entropy": 0.061232382198795676, "clip_ratio/low_mean": 0.005462184897623956, "clip_ratio/low_min": 0.005462184897623956, "clip_ratio/high_mean": 0.0017361111240461469, "clip_ratio/high_max": 0.0017361111240461469, "clip_ratio/region_mean": 0.007198296021670103, "reward_total_mean": 0.3661602735519409, "reward_meter_mean": 0.7864480018615723, "reward_meter_std": 0.26775890588760376, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5, "reward_repeat_penalty_std": 0.1511857956647873, "reward_total_composite_mean": 0.3661602735519409, "reward_total_composite_std": 0.13910025358200073} {"timestamp_utc": "2026-04-11T23:36:56Z", "mode": "train", "global_step": 1122, "epoch": 0.0450656705627184, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.603030303030303e-06, "num_tokens": 2530984.0, "completions/mean_length": 73.0, "completions/min_length": 73.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.0, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9929870963096619, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9929870963096619, "rewards/total_composite/std": 0.0, "reward": 0.9929870963096619, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0027792660985141993, "sampling/sampling_logp_difference/max": 0.229114830493927, "sampling/importance_sampling_ratio/min": 0.7952372431755066, "sampling/importance_sampling_ratio/mean": 1.000472068786621, "sampling/importance_sampling_ratio/max": 1.0893265008926392, "entropy": 0.021201700437813997, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9929870963096619, "reward_meter_mean": 0.9929870963096619, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9929870963096619, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-11T23:37:01Z", "mode": "train", "global_step": 1123, "epoch": 0.045105836044503356, "loss": -0.0032, "grad_norm": 4.8726959228515625, "learning_rate": 6.600000000000001e-06, "num_tokens": 2532779.0, "completions/mean_length": 66.375, "completions/min_length": 65.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.375, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.996692419052124, "rewards/meter/std": 0.0006456730188801885, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.996692419052124, "rewards/total_composite/std": 0.0006456730188801885, "reward": 0.996692419052124, "reward_std": 0.0006456732517108321, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04234900325536728, "sampling/sampling_logp_difference/max": 1.778670310974121, "sampling/importance_sampling_ratio/min": 0.16886253654956818, "sampling/importance_sampling_ratio/mean": 0.9999663233757019, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.198346434161067, "clip_ratio/low_mean": 0.01133623847272247, "clip_ratio/low_min": 0.01133623847272247, "clip_ratio/high_mean": 0.0149312699213624, "clip_ratio/high_max": 0.0149312699213624, "clip_ratio/region_mean": 0.02626750839408487, "reward_total_mean": 0.996692419052124, "reward_meter_mean": 0.996692419052124, "reward_meter_std": 0.0006456730188801885, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.996692419052124, "reward_total_composite_std": 0.0006456730188801885} {"timestamp_utc": "2026-04-11T23:37:05Z", "mode": "train", "global_step": 1124, "epoch": 0.04514600152628831, "loss": 0.0006, "grad_norm": 0.6933730244636536, "learning_rate": 6.596969696969698e-06, "num_tokens": 2534594.0, "completions/mean_length": 72.875, "completions/min_length": 72.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.875, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9930577874183655, "rewards/meter/std": 0.0013781489105895162, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9930577874183655, "rewards/total_composite/std": 0.0013781489105895162, "reward": 0.9930577874183655, "reward_std": 0.001378138200379908, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004648053552955389, "sampling/sampling_logp_difference/max": 0.48731809854507446, "sampling/importance_sampling_ratio/min": 0.6142716407775879, "sampling/importance_sampling_ratio/mean": 1.000672459602356, "sampling/importance_sampling_ratio/max": 1.413206696510315, "entropy": 0.028838476166129112, "clip_ratio/low_mean": 0.006849315017461777, "clip_ratio/low_min": 0.006849315017461777, "clip_ratio/high_mean": 0.0034722222480922937, "clip_ratio/high_max": 0.0034722222480922937, "clip_ratio/region_mean": 0.01032153726555407, "reward_total_mean": 0.9930577874183655, "reward_meter_mean": 0.9930577874183655, "reward_meter_std": 0.0013781489105895162, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9930577874183655, "reward_total_composite_std": 0.0013781489105895162} {"timestamp_utc": "2026-04-11T23:37:11Z", "mode": "train", "global_step": 1125, "epoch": 0.045186167008073264, "loss": -0.0499, "grad_norm": 3.602660894393921, "learning_rate": 6.593939393939395e-06, "num_tokens": 2536357.0, "completions/mean_length": 70.375, "completions/min_length": 61.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.375, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9664039611816406, "rewards/meter/std": 0.08127888292074203, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9664039611816406, "rewards/total_composite/std": 0.08127888292074203, "reward": 0.9664039611816406, "reward_std": 0.08127887547016144, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015036996454000473, "sampling/sampling_logp_difference/max": 1.4699931144714355, "sampling/importance_sampling_ratio/min": 0.22992707788944244, "sampling/importance_sampling_ratio/mean": 0.9998224377632141, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.051455921959131956, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.010517970658838749, "clip_ratio/high_max": 0.010517970658838749, "clip_ratio/region_mean": 0.010517970658838749, "reward_total_mean": 0.9664039611816406, "reward_meter_mean": 0.9664039611816406, "reward_meter_std": 0.08127888292074203, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9664039611816406, "reward_total_composite_std": 0.08127888292074203} {"timestamp_utc": "2026-04-11T23:37:16Z", "mode": "train", "global_step": 1126, "epoch": 0.04522633248985822, "loss": -0.0067, "grad_norm": 4.485559940338135, "learning_rate": 6.590909090909091e-06, "num_tokens": 2538137.0, "completions/mean_length": 66.5, "completions/min_length": 65.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.5, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9959009289741516, "rewards/meter/std": 0.0018093610415235162, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9959009289741516, "rewards/total_composite/std": 0.0018093610415235162, "reward": 0.9959009289741516, "reward_std": 0.0018093656981363893, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030639950186014175, "sampling/sampling_logp_difference/max": 1.1065802574157715, "sampling/importance_sampling_ratio/min": 0.33068788051605225, "sampling/importance_sampling_ratio/mean": 0.997355580329895, "sampling/importance_sampling_ratio/max": 1.4944887161254883, "entropy": 0.15226891916245222, "clip_ratio/low_mean": 0.007549191126599908, "clip_ratio/low_min": 0.007549191126599908, "clip_ratio/high_mean": 0.02241882192902267, "clip_ratio/high_max": 0.02241882192902267, "clip_ratio/region_mean": 0.029968013055622578, "reward_total_mean": 0.9959009289741516, "reward_meter_mean": 0.9959009289741516, "reward_meter_std": 0.0018093610415235162, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9959009289741516, "reward_total_composite_std": 0.0018093610415235162} {"timestamp_utc": "2026-04-11T23:37:20Z", "mode": "train", "global_step": 1127, "epoch": 0.04526649797164317, "loss": -0.0117, "grad_norm": 3.99946928024292, "learning_rate": 6.5878787878787885e-06, "num_tokens": 2539806.0, "completions/mean_length": 60.625, "completions/min_length": 58.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.625, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9501904249191284, "rewards/meter/std": 0.06670597940683365, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9501904249191284, "rewards/total_composite/std": 0.06670597940683365, "reward": 0.9501904249191284, "reward_std": 0.06670597940683365, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022387670353055, "sampling/sampling_logp_difference/max": 1.423022985458374, "sampling/importance_sampling_ratio/min": 0.24098443984985352, "sampling/importance_sampling_ratio/mean": 0.9991652369499207, "sampling/importance_sampling_ratio/max": 1.6531167030334473, "entropy": 0.08485704753547907, "clip_ratio/low_mean": 0.006355932215228677, "clip_ratio/low_min": 0.006355932215228677, "clip_ratio/high_mean": 0.01383362547494471, "clip_ratio/high_max": 0.01383362547494471, "clip_ratio/region_mean": 0.020189557690173388, "reward_total_mean": 0.9501904249191284, "reward_meter_mean": 0.9501904249191284, "reward_meter_std": 0.06670597940683365, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9501904249191284, "reward_total_composite_std": 0.06670597940683365} {"timestamp_utc": "2026-04-11T23:37:25Z", "mode": "train", "global_step": 1128, "epoch": 0.045306663453428125, "loss": -0.0059, "grad_norm": 1.894775390625, "learning_rate": 6.584848484848485e-06, "num_tokens": 2541433.0, "completions/mean_length": 61.375, "completions/min_length": 60.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.375, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9959591627120972, "rewards/meter/std": 0.0002977726107928902, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9959591627120972, "rewards/total_composite/std": 0.0002977726107928902, "reward": 0.9959591627120972, "reward_std": 0.0002977752883452922, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008953627198934555, "sampling/sampling_logp_difference/max": 0.6550607681274414, "sampling/importance_sampling_ratio/min": 0.5194104909896851, "sampling/importance_sampling_ratio/mean": 1.0015958547592163, "sampling/importance_sampling_ratio/max": 1.208601951599121, "entropy": 0.05297265062108636, "clip_ratio/low_mean": 0.0020833334419876337, "clip_ratio/low_min": 0.0020833334419876337, "clip_ratio/high_mean": 0.006048386916518211, "clip_ratio/high_max": 0.006048386916518211, "clip_ratio/region_mean": 0.008131720358505845, "reward_total_mean": 0.9959591627120972, "reward_meter_mean": 0.9959591627120972, "reward_meter_std": 0.0002977726107928902, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9959591627120972, "reward_total_composite_std": 0.0002977726107928902} {"timestamp_utc": "2026-04-11T23:37:35Z", "mode": "train", "global_step": 1129, "epoch": 0.04534682893521308, "loss": -0.2408, "grad_norm": 1.491158366203308, "learning_rate": 6.581818181818182e-06, "num_tokens": 2544019.0, "completions/mean_length": 375.25, "completions/min_length": 221.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.5, "completions/mean_terminated_length": 238.5, "completions/min_terminated_length": 221.0, "completions/max_terminated_length": 277.0, "rewards/meter/mean": 0.367766410112381, "rewards/meter/std": 0.4184606671333313, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.2777460217475891, "rewards/arabic_clean/mean": 0.5, "rewards/arabic_clean/std": 0.5345224738121033, "rewards/repeat_penalty/mean": 0.7755848169326782, "rewards/repeat_penalty/std": 0.2187419831752777, "rewards/total_composite/mean": 0.16618868708610535, "rewards/total_composite/std": 0.2396627515554428, "reward": 0.16618868708610535, "reward_std": 0.2396627515554428, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03679489716887474, "sampling/sampling_logp_difference/max": 5.3318939208984375, "sampling/importance_sampling_ratio/min": 0.004834904335439205, "sampling/importance_sampling_ratio/mean": 1.0069185495376587, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09304305911064148, "clip_ratio/low_mean": 0.005044843070209026, "clip_ratio/low_min": 0.005044843070209026, "clip_ratio/high_mean": 0.0071450776886194944, "clip_ratio/high_max": 0.0071450776886194944, "clip_ratio/region_mean": 0.01218992075882852, "reward_total_mean": 0.16618868708610535, "reward_meter_mean": 0.367766410112381, "reward_meter_std": 0.4184606671333313, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.2777460217475891, "reward_arabic_clean_mean": 0.5, "reward_arabic_clean_std": 0.5345224738121033, "reward_repeat_penalty_mean": 0.7755848169326782, "reward_repeat_penalty_std": 0.2187419831752777, "reward_total_composite_mean": 0.16618868708610535, "reward_total_composite_std": 0.2396627515554428} {"timestamp_utc": "2026-04-11T23:37:44Z", "mode": "train", "global_step": 1130, "epoch": 0.04538699441699803, "loss": -0.0873, "grad_norm": 3.4096953868865967, "learning_rate": 6.578787878787879e-06, "num_tokens": 2545823.0, "completions/mean_length": 108.5, "completions/min_length": 46.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 50.85714340209961, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.5511125326156616, "rewards/meter/std": 0.42126619815826416, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.47555798292160034, "rewards/total_composite/std": 0.38399630784988403, "reward": 0.47555798292160034, "reward_std": 0.38399630784988403, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.054451193660497665, "sampling/sampling_logp_difference/max": 1.399144172668457, "sampling/importance_sampling_ratio/min": 0.24680811166763306, "sampling/importance_sampling_ratio/mean": 1.007746696472168, "sampling/importance_sampling_ratio/max": 1.7676554918289185, "entropy": 0.25371737964451313, "clip_ratio/low_mean": 0.015434782486408949, "clip_ratio/low_min": 0.015434782486408949, "clip_ratio/high_mean": 0.030356566421687603, "clip_ratio/high_max": 0.030356566421687603, "clip_ratio/region_mean": 0.04579134890809655, "reward_total_mean": 0.47555798292160034, "reward_meter_mean": 0.5511125326156616, "reward_meter_std": 0.42126619815826416, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.17251639068126678, "reward_total_composite_mean": 0.47555798292160034, "reward_total_composite_std": 0.38399630784988403} {"timestamp_utc": "2026-04-11T23:37:49Z", "mode": "train", "global_step": 1131, "epoch": 0.04542715989878299, "loss": -0.0297, "grad_norm": 8.209676742553711, "learning_rate": 6.575757575757577e-06, "num_tokens": 2547552.0, "completions/mean_length": 54.125, "completions/min_length": 50.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.125, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9417009353637695, "rewards/meter/std": 0.008068003691732883, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9417009353637695, "rewards/total_composite/std": 0.008068003691732883, "reward": 0.9417009353637695, "reward_std": 0.008068017661571503, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02913379855453968, "sampling/sampling_logp_difference/max": 2.463469982147217, "sampling/importance_sampling_ratio/min": 0.08513900637626648, "sampling/importance_sampling_ratio/mean": 0.9954664707183838, "sampling/importance_sampling_ratio/max": 1.376886010169983, "entropy": 0.08877178048714995, "clip_ratio/low_mean": 0.012500000186264515, "clip_ratio/low_min": 0.012500000186264515, "clip_ratio/high_mean": 0.013678450835868716, "clip_ratio/high_max": 0.013678450835868716, "clip_ratio/region_mean": 0.02617845102213323, "reward_total_mean": 0.9417009353637695, "reward_meter_mean": 0.9417009353637695, "reward_meter_std": 0.008068003691732883, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9417009353637695, "reward_total_composite_std": 0.008068003691732883} {"timestamp_utc": "2026-04-11T23:37:53Z", "mode": "train", "global_step": 1132, "epoch": 0.04546732538056794, "loss": -0.0471, "grad_norm": 14.512413024902344, "learning_rate": 6.572727272727273e-06, "num_tokens": 2548978.0, "completions/mean_length": 29.25, "completions/min_length": 26.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.25, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.8743196725845337, "rewards/meter/std": 0.20709700882434845, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8743196725845337, "rewards/total_composite/std": 0.20709700882434845, "reward": 0.8743196725845337, "reward_std": 0.20709700882434845, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09784162044525146, "sampling/sampling_logp_difference/max": 6.468225002288818, "sampling/importance_sampling_ratio/min": 0.0015519780572503805, "sampling/importance_sampling_ratio/mean": 1.0037531852722168, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.42541524581611156, "clip_ratio/low_mean": 0.014423077460378408, "clip_ratio/low_min": 0.014423077460378408, "clip_ratio/high_mean": 0.05365684116259217, "clip_ratio/high_max": 0.05365684116259217, "clip_ratio/region_mean": 0.06807991862297058, "reward_total_mean": 0.8743196725845337, "reward_meter_mean": 0.8743196725845337, "reward_meter_std": 0.20709700882434845, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8743196725845337, "reward_total_composite_std": 0.20709700882434845} {"timestamp_utc": "2026-04-11T23:37:58Z", "mode": "train", "global_step": 1133, "epoch": 0.045507490862352895, "loss": 0.0098, "grad_norm": 1.9009183645248413, "learning_rate": 6.56969696969697e-06, "num_tokens": 2550664.0, "completions/mean_length": 59.75, "completions/min_length": 59.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.75, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9870603084564209, "rewards/meter/std": 0.004516107961535454, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9870603084564209, "rewards/total_composite/std": 0.004516107961535454, "reward": 0.9870603084564209, "reward_std": 0.004516112618148327, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013605736196041107, "sampling/sampling_logp_difference/max": 1.056100606918335, "sampling/importance_sampling_ratio/min": 0.4231666624546051, "sampling/importance_sampling_ratio/mean": 1.0032687187194824, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.046739939134567976, "clip_ratio/low_mean": 0.006215847097337246, "clip_ratio/low_min": 0.006215847097337246, "clip_ratio/high_mean": 0.004167824285104871, "clip_ratio/high_max": 0.004167824285104871, "clip_ratio/region_mean": 0.010383671382442117, "reward_total_mean": 0.9870603084564209, "reward_meter_mean": 0.9870603084564209, "reward_meter_std": 0.004516107961535454, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9870603084564209, "reward_total_composite_std": 0.004516107961535454} {"timestamp_utc": "2026-04-11T23:38:03Z", "mode": "train", "global_step": 1134, "epoch": 0.04554765634413785, "loss": 0.0062, "grad_norm": 4.714395523071289, "learning_rate": 6.566666666666667e-06, "num_tokens": 2552346.0, "completions/mean_length": 70.25, "completions/min_length": 68.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.25, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9971396923065186, "rewards/meter/std": 0.0017616376280784607, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9971396923065186, "rewards/total_composite/std": 0.0017616376280784607, "reward": 0.9971396923065186, "reward_std": 0.0017616377444937825, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030578911304473877, "sampling/sampling_logp_difference/max": 1.3415107727050781, "sampling/importance_sampling_ratio/min": 0.2614503800868988, "sampling/importance_sampling_ratio/mean": 1.0078562498092651, "sampling/importance_sampling_ratio/max": 1.841591238975525, "entropy": 0.18766964972019196, "clip_ratio/low_mean": 0.010638833045959473, "clip_ratio/low_min": 0.010638833045959473, "clip_ratio/high_mean": 0.026770169381052256, "clip_ratio/high_max": 0.026770169381052256, "clip_ratio/region_mean": 0.03740900242701173, "reward_total_mean": 0.9971396923065186, "reward_meter_mean": 0.9971396923065186, "reward_meter_std": 0.0017616376280784607, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9971396923065186, "reward_total_composite_std": 0.0017616376280784607} {"timestamp_utc": "2026-04-11T23:38:08Z", "mode": "train", "global_step": 1135, "epoch": 0.0455878218259228, "loss": 0.0005, "grad_norm": 3.5454463958740234, "learning_rate": 6.563636363636364e-06, "num_tokens": 2554294.0, "completions/mean_length": 76.5, "completions/min_length": 73.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.5, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9800901412963867, "rewards/meter/std": 0.01382446475327015, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9800901412963867, "rewards/total_composite/std": 0.01382446475327015, "reward": 0.9800901412963867, "reward_std": 0.013824466615915298, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01574569195508957, "sampling/sampling_logp_difference/max": 0.8164339065551758, "sampling/importance_sampling_ratio/min": 0.44200506806373596, "sampling/importance_sampling_ratio/mean": 1.005315899848938, "sampling/importance_sampling_ratio/max": 1.7477397918701172, "entropy": 0.10036304593086243, "clip_ratio/low_mean": 0.00811688310932368, "clip_ratio/low_min": 0.00811688310932368, "clip_ratio/high_mean": 0.011680196272209287, "clip_ratio/high_max": 0.011680196272209287, "clip_ratio/region_mean": 0.019797079381532967, "reward_total_mean": 0.9800901412963867, "reward_meter_mean": 0.9800901412963867, "reward_meter_std": 0.01382446475327015, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9800901412963867, "reward_total_composite_std": 0.01382446475327015} {"timestamp_utc": "2026-04-11T23:38:17Z", "mode": "train", "global_step": 1136, "epoch": 0.04562798730770776, "loss": -0.019, "grad_norm": 2.327674627304077, "learning_rate": 6.56060606060606e-06, "num_tokens": 2559265.0, "completions/mean_length": 388.375, "completions/min_length": 362.0, "completions/max_length": 427.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 388.375, "completions/min_terminated_length": 362.0, "completions/max_terminated_length": 427.0, "rewards/meter/mean": 0.952599823474884, "rewards/meter/std": 0.09515554457902908, "rewards/count_adherence/mean": 0.7410714626312256, "rewards/count_adherence/std": 0.03696778044104576, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.716478705406189, "rewards/repeat_penalty/std": 0.12055735290050507, "rewards/total_composite/mean": 0.49994567036628723, "rewards/total_composite/std": 0.06088562682271004, "reward": 0.49994567036628723, "reward_std": 0.060885630548000336, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02755453996360302, "sampling/sampling_logp_difference/max": 1.6922149658203125, "sampling/importance_sampling_ratio/min": 0.18411126732826233, "sampling/importance_sampling_ratio/mean": 1.0076520442962646, "sampling/importance_sampling_ratio/max": 1.8240277767181396, "entropy": 0.21599608566612005, "clip_ratio/low_mean": 0.010929276293609291, "clip_ratio/low_min": 0.010929276293609291, "clip_ratio/high_mean": 0.005753153818659484, "clip_ratio/high_max": 0.005753153818659484, "clip_ratio/region_mean": 0.016682430112268776, "reward_total_mean": 0.49994567036628723, "reward_meter_mean": 0.952599823474884, "reward_meter_std": 0.09515554457902908, "reward_count_adherence_mean": 0.7410714626312256, "reward_count_adherence_std": 0.03696778044104576, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.716478705406189, "reward_repeat_penalty_std": 0.12055735290050507, "reward_total_composite_mean": 0.49994567036628723, "reward_total_composite_std": 0.06088562682271004} {"timestamp_utc": "2026-04-11T23:38:23Z", "mode": "train", "global_step": 1137, "epoch": 0.04566815278949271, "loss": 0.0836, "grad_norm": 3.7329790592193604, "learning_rate": 6.5575757575757585e-06, "num_tokens": 2561586.0, "completions/mean_length": 111.125, "completions/min_length": 100.0, "completions/max_length": 138.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 111.125, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.9837856292724609, "rewards/meter/std": 0.012296498753130436, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7892857789993286, "rewards/repeat_penalty/std": 0.030304575338959694, "rewards/total_composite/mean": 0.7466674447059631, "rewards/total_composite/std": 0.11019133776426315, "reward": 0.7466674447059631, "reward_std": 0.11019132286310196, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020305845886468887, "sampling/sampling_logp_difference/max": 2.0209195613861084, "sampling/importance_sampling_ratio/min": 0.13253353536128998, "sampling/importance_sampling_ratio/mean": 0.9983324408531189, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08285832032561302, "clip_ratio/low_mean": 0.0009057971183210611, "clip_ratio/low_min": 0.0009057971183210611, "clip_ratio/high_mean": 0.010732458787970245, "clip_ratio/high_max": 0.010732458787970245, "clip_ratio/region_mean": 0.011638255906291306, "reward_total_mean": 0.7466674447059631, "reward_meter_mean": 0.9837856292724609, "reward_meter_std": 0.012296498753130436, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7892857789993286, "reward_repeat_penalty_std": 0.030304575338959694, "reward_total_composite_mean": 0.7466674447059631, "reward_total_composite_std": 0.11019133776426315} {"timestamp_utc": "2026-04-11T23:38:28Z", "mode": "train", "global_step": 1138, "epoch": 0.045708318271277665, "loss": -0.0078, "grad_norm": 2.7455623149871826, "learning_rate": 6.554545454545455e-06, "num_tokens": 2563284.0, "completions/mean_length": 59.25, "completions/min_length": 59.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.25, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9892421960830688, "rewards/meter/std": 0.0002119775745086372, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9892421960830688, "rewards/total_composite/std": 0.0002119775745086372, "reward": 0.9892421960830688, "reward_std": 0.000211986611247994, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.002105327555909753, "sampling/sampling_logp_difference/max": 0.09289813041687012, "sampling/importance_sampling_ratio/min": 0.960770845413208, "sampling/importance_sampling_ratio/mean": 1.0011262893676758, "sampling/importance_sampling_ratio/max": 1.097350001335144, "entropy": 0.020128208212554455, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9892421960830688, "reward_meter_mean": 0.9892421960830688, "reward_meter_std": 0.0002119775745086372, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9892421960830688, "reward_total_composite_std": 0.0002119775745086372} {"timestamp_utc": "2026-04-11T23:38:33Z", "mode": "train", "global_step": 1139, "epoch": 0.04574848375306262, "loss": -0.0093, "grad_norm": 3.7970211505889893, "learning_rate": 6.551515151515152e-06, "num_tokens": 2564995.0, "completions/mean_length": 60.875, "completions/min_length": 59.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.875, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.989832878112793, "rewards/meter/std": 0.0006110123940743506, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.989832878112793, "rewards/total_composite/std": 0.0006110123940743506, "reward": 0.989832878112793, "reward_std": 0.0006110109388828278, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011983020231127739, "sampling/sampling_logp_difference/max": 1.2451696395874023, "sampling/importance_sampling_ratio/min": 0.28789207339286804, "sampling/importance_sampling_ratio/mean": 1.0028979778289795, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.04918731888756156, "clip_ratio/low_mean": 0.008266184711828828, "clip_ratio/low_min": 0.008266184711828828, "clip_ratio/high_mean": 0.002016128972172737, "clip_ratio/high_max": 0.002016128972172737, "clip_ratio/region_mean": 0.010282313684001565, "reward_total_mean": 0.989832878112793, "reward_meter_mean": 0.989832878112793, "reward_meter_std": 0.0006110123940743506, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.989832878112793, "reward_total_composite_std": 0.0006110123940743506} {"timestamp_utc": "2026-04-11T23:38:38Z", "mode": "train", "global_step": 1140, "epoch": 0.04578864923484757, "loss": -0.0165, "grad_norm": 5.715131759643555, "learning_rate": 6.5484848484848494e-06, "num_tokens": 2566645.0, "completions/mean_length": 56.25, "completions/min_length": 54.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.25, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.2350359857082367, "rewards/meter/std": 0.07872093468904495, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.2350359857082367, "rewards/total_composite/std": 0.07872093468904495, "reward": 0.2350359857082367, "reward_std": 0.07872093468904495, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012172389775514603, "sampling/sampling_logp_difference/max": 0.5825954675674438, "sampling/importance_sampling_ratio/min": 0.6520004272460938, "sampling/importance_sampling_ratio/mean": 1.001919150352478, "sampling/importance_sampling_ratio/max": 1.7906800508499146, "entropy": 0.061810399405658245, "clip_ratio/low_mean": 0.002314814832061529, "clip_ratio/low_min": 0.002314814832061529, "clip_ratio/high_mean": 0.00657894741743803, "clip_ratio/high_max": 0.00657894741743803, "clip_ratio/region_mean": 0.00889376224949956, "reward_total_mean": 0.2350359857082367, "reward_meter_mean": 0.2350359857082367, "reward_meter_std": 0.07872093468904495, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.2350359857082367, "reward_total_composite_std": 0.07872093468904495} {"timestamp_utc": "2026-04-11T23:38:43Z", "mode": "train", "global_step": 1141, "epoch": 0.04582881471663253, "loss": 0.0143, "grad_norm": 7.939047813415527, "learning_rate": 6.545454545454546e-06, "num_tokens": 2568290.0, "completions/mean_length": 53.625, "completions/min_length": 53.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.625, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9498906135559082, "rewards/meter/std": 0.007521784398704767, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9498906135559082, "rewards/total_composite/std": 0.007521784398704767, "reward": 0.9498906135559082, "reward_std": 0.007521784398704767, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04453643411397934, "sampling/sampling_logp_difference/max": 2.4130172729492188, "sampling/importance_sampling_ratio/min": 0.08954470604658127, "sampling/importance_sampling_ratio/mean": 0.9988313913345337, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1192027423530817, "clip_ratio/low_mean": 0.016250555869191885, "clip_ratio/low_min": 0.016250555869191885, "clip_ratio/high_mean": 0.02358490601181984, "clip_ratio/high_max": 0.02358490601181984, "clip_ratio/region_mean": 0.039835461881011724, "reward_total_mean": 0.9498906135559082, "reward_meter_mean": 0.9498906135559082, "reward_meter_std": 0.007521784398704767, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9498906135559082, "reward_total_composite_std": 0.007521784398704767} {"timestamp_utc": "2026-04-11T23:38:48Z", "mode": "train", "global_step": 1142, "epoch": 0.04586898019841748, "loss": 0.0038, "grad_norm": 4.282461166381836, "learning_rate": 6.542424242424243e-06, "num_tokens": 2569755.0, "completions/mean_length": 39.125, "completions/min_length": 37.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.125, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.9843659400939941, "rewards/meter/std": 0.01807660609483719, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9843659400939941, "rewards/total_composite/std": 0.01807660609483719, "reward": 0.9843659400939941, "reward_std": 0.018076593056321144, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03325369209051132, "sampling/sampling_logp_difference/max": 1.2249364852905273, "sampling/importance_sampling_ratio/min": 0.29377633333206177, "sampling/importance_sampling_ratio/mean": 0.9939477443695068, "sampling/importance_sampling_ratio/max": 1.3938076496124268, "entropy": 0.17795206978917122, "clip_ratio/low_mean": 0.0031250000465661287, "clip_ratio/low_min": 0.0031250000465661287, "clip_ratio/high_mean": 0.025911615695804358, "clip_ratio/high_max": 0.025911615695804358, "clip_ratio/region_mean": 0.029036615742370486, "reward_total_mean": 0.9843659400939941, "reward_meter_mean": 0.9843659400939941, "reward_meter_std": 0.01807660609483719, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9843659400939941, "reward_total_composite_std": 0.01807660609483719} {"timestamp_utc": "2026-04-11T23:38:58Z", "mode": "train", "global_step": 1143, "epoch": 0.045909145680202434, "loss": -0.0509, "grad_norm": 1.155205488204956, "learning_rate": 6.5393939393939395e-06, "num_tokens": 2575500.0, "completions/mean_length": 441.125, "completions/min_length": 400.0, "completions/max_length": 498.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 441.125, "completions/min_terminated_length": 400.0, "completions/max_terminated_length": 498.0, "rewards/meter/mean": 0.9610804319381714, "rewards/meter/std": 0.050803039222955704, "rewards/count_adherence/mean": 0.6323529481887817, "rewards/count_adherence/std": 0.027230001986026764, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5833333134651184, "rewards/repeat_penalty/std": 0.19633837044239044, "rewards/total_composite/mean": 0.35644832253456116, "rewards/total_composite/std": 0.13268794119358063, "reward": 0.35644832253456116, "reward_std": 0.13268794119358063, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021479658782482147, "sampling/sampling_logp_difference/max": 1.2659144401550293, "sampling/importance_sampling_ratio/min": 0.28720539808273315, "sampling/importance_sampling_ratio/mean": 1.004508376121521, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1385105513036251, "clip_ratio/low_mean": 0.004563539958326146, "clip_ratio/low_min": 0.004563539958326146, "clip_ratio/high_mean": 0.011979256058111787, "clip_ratio/high_max": 0.011979256058111787, "clip_ratio/region_mean": 0.016542796016437933, "reward_total_mean": 0.35644832253456116, "reward_meter_mean": 0.9610804319381714, "reward_meter_std": 0.050803039222955704, "reward_count_adherence_mean": 0.6323529481887817, "reward_count_adherence_std": 0.027230001986026764, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5833333134651184, "reward_repeat_penalty_std": 0.19633837044239044, "reward_total_composite_mean": 0.35644832253456116, "reward_total_composite_std": 0.13268794119358063} {"timestamp_utc": "2026-04-11T23:39:08Z", "mode": "train", "global_step": 1144, "epoch": 0.04594931116198739, "loss": -0.0143, "grad_norm": 0.9518696665763855, "learning_rate": 6.536363636363638e-06, "num_tokens": 2580194.0, "completions/mean_length": 387.75, "completions/min_length": 354.0, "completions/max_length": 420.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 387.75, "completions/min_terminated_length": 354.0, "completions/max_terminated_length": 420.0, "rewards/meter/mean": 0.9527665376663208, "rewards/meter/std": 0.0463380366563797, "rewards/count_adherence/mean": 0.7053571939468384, "rewards/count_adherence/std": 0.0707879438996315, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5333124399185181, "rewards/repeat_penalty/std": 0.15502820909023285, "rewards/total_composite/mean": 0.36780261993408203, "rewards/total_composite/std": 0.1357763111591339, "reward": 0.36780261993408203, "reward_std": 0.1357763111591339, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011555654928088188, "sampling/sampling_logp_difference/max": 3.7675018310546875, "sampling/importance_sampling_ratio/min": 0.02310972288250923, "sampling/importance_sampling_ratio/mean": 1.0018688440322876, "sampling/importance_sampling_ratio/max": 1.8800609111785889, "entropy": 0.06274456530809402, "clip_ratio/low_mean": 0.0020058397494722158, "clip_ratio/low_min": 0.0020058397494722158, "clip_ratio/high_mean": 0.005727101757656783, "clip_ratio/high_max": 0.005727101757656783, "clip_ratio/region_mean": 0.007732941507128999, "reward_total_mean": 0.36780261993408203, "reward_meter_mean": 0.9527665376663208, "reward_meter_std": 0.0463380366563797, "reward_count_adherence_mean": 0.7053571939468384, "reward_count_adherence_std": 0.0707879438996315, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5333124399185181, "reward_repeat_penalty_std": 0.15502820909023285, "reward_total_composite_mean": 0.36780261993408203, "reward_total_composite_std": 0.1357763111591339} {"timestamp_utc": "2026-04-11T23:39:13Z", "mode": "train", "global_step": 1145, "epoch": 0.04598947664377234, "loss": -0.0246, "grad_norm": 3.6901073455810547, "learning_rate": 6.533333333333334e-06, "num_tokens": 2582143.0, "completions/mean_length": 77.625, "completions/min_length": 75.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.625, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9917942881584167, "rewards/meter/std": 0.0048305438831448555, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9917942881584167, "rewards/total_composite/std": 0.0048305438831448555, "reward": 0.9917942881584167, "reward_std": 0.004830518271774054, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017006773501634598, "sampling/sampling_logp_difference/max": 1.3061323165893555, "sampling/importance_sampling_ratio/min": 0.34659311175346375, "sampling/importance_sampling_ratio/mean": 1.0097332000732422, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.13493561558425426, "clip_ratio/low_mean": 0.006580086657777429, "clip_ratio/low_min": 0.006580086657777429, "clip_ratio/high_mean": 0.0031250000465661287, "clip_ratio/high_max": 0.0031250000465661287, "clip_ratio/region_mean": 0.009705086704343557, "reward_total_mean": 0.9917942881584167, "reward_meter_mean": 0.9917942881584167, "reward_meter_std": 0.0048305438831448555, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9917942881584167, "reward_total_composite_std": 0.0048305438831448555} {"timestamp_utc": "2026-04-11T23:39:18Z", "mode": "train", "global_step": 1146, "epoch": 0.046029642125557296, "loss": -0.0049, "grad_norm": 1.9219002723693848, "learning_rate": 6.530303030303031e-06, "num_tokens": 2583831.0, "completions/mean_length": 72.0, "completions/min_length": 71.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.0, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9963383078575134, "rewards/meter/std": 0.0003733669000212103, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9548450708389282, "rewards/total_composite/std": 0.11756306886672974, "reward": 0.9548450708389282, "reward_std": 0.11756306141614914, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003385355230420828, "sampling/sampling_logp_difference/max": 0.539313793182373, "sampling/importance_sampling_ratio/min": 0.5831482410430908, "sampling/importance_sampling_ratio/mean": 1.000481128692627, "sampling/importance_sampling_ratio/max": 1.0752372741699219, "entropy": 0.016070799436420202, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9548450708389282, "reward_meter_mean": 0.9963383078575134, "reward_meter_std": 0.0003733669000212103, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.9548450708389282, "reward_total_composite_std": 0.11756306886672974} {"timestamp_utc": "2026-04-11T23:39:24Z", "mode": "train", "global_step": 1147, "epoch": 0.04606980760734225, "loss": 0.0294, "grad_norm": 15.72409725189209, "learning_rate": 6.527272727272728e-06, "num_tokens": 2585484.0, "completions/mean_length": 32.625, "completions/min_length": 30.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.625, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.89949631690979, "rewards/meter/std": 0.18807001411914825, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.89949631690979, "rewards/total_composite/std": 0.18807001411914825, "reward": 0.89949631690979, "reward_std": 0.18807001411914825, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06001806631684303, "sampling/sampling_logp_difference/max": 1.7198677062988281, "sampling/importance_sampling_ratio/min": 0.29962798953056335, "sampling/importance_sampling_ratio/mean": 1.0089141130447388, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.21602893620729446, "clip_ratio/low_mean": 0.01907169120386243, "clip_ratio/low_min": 0.01907169120386243, "clip_ratio/high_mean": 0.014285714365541935, "clip_ratio/high_max": 0.014285714365541935, "clip_ratio/region_mean": 0.033357405569404364, "reward_total_mean": 0.89949631690979, "reward_meter_mean": 0.89949631690979, "reward_meter_std": 0.18807001411914825, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.89949631690979, "reward_total_composite_std": 0.18807001411914825} {"timestamp_utc": "2026-04-11T23:39:28Z", "mode": "train", "global_step": 1148, "epoch": 0.046109973089127204, "loss": 0.0136, "grad_norm": 5.76182746887207, "learning_rate": 6.524242424242425e-06, "num_tokens": 2586905.0, "completions/mean_length": 37.625, "completions/min_length": 37.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.625, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.9881135821342468, "rewards/meter/std": 0.0235806442797184, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9881135821342468, "rewards/total_composite/std": 0.0235806442797184, "reward": 0.9881135821342468, "reward_std": 0.02358064614236355, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010363386012613773, "sampling/sampling_logp_difference/max": 0.48454809188842773, "sampling/importance_sampling_ratio/min": 0.6159754991531372, "sampling/importance_sampling_ratio/mean": 1.0021705627441406, "sampling/importance_sampling_ratio/max": 1.4336398839950562, "entropy": 0.03891826095059514, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.00657894741743803, "clip_ratio/high_max": 0.00657894741743803, "clip_ratio/region_mean": 0.00657894741743803, "reward_total_mean": 0.9881135821342468, "reward_meter_mean": 0.9881135821342468, "reward_meter_std": 0.0235806442797184, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9881135821342468, "reward_total_composite_std": 0.0235806442797184} {"timestamp_utc": "2026-04-11T23:39:33Z", "mode": "train", "global_step": 1149, "epoch": 0.04615013857091216, "loss": 0.0133, "grad_norm": 2.983004331588745, "learning_rate": 6.521212121212121e-06, "num_tokens": 2588889.0, "completions/mean_length": 77.0, "completions/min_length": 71.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.0, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9948194026947021, "rewards/meter/std": 0.007181616500020027, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.9116114377975464, "rewards/total_composite/std": 0.1519550234079361, "reward": 0.9116114377975464, "reward_std": 0.1519550234079361, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022046150639653206, "sampling/sampling_logp_difference/max": 0.7371308207511902, "sampling/importance_sampling_ratio/min": 0.4784848093986511, "sampling/importance_sampling_ratio/mean": 1.003957748413086, "sampling/importance_sampling_ratio/max": 1.373502492904663, "entropy": 0.11253493744879961, "clip_ratio/low_mean": 0.011157091706991196, "clip_ratio/low_min": 0.011157091706991196, "clip_ratio/high_mean": 0.021323838154785335, "clip_ratio/high_max": 0.021323838154785335, "clip_ratio/region_mean": 0.03248092986177653, "reward_total_mean": 0.9116114377975464, "reward_meter_mean": 0.9948194026947021, "reward_meter_std": 0.007181616500020027, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_total_composite_mean": 0.9116114377975464, "reward_total_composite_std": 0.1519550234079361} {"timestamp_utc": "2026-04-11T23:39:40Z", "mode": "train", "global_step": 1150, "epoch": 0.04619030405269711, "loss": -0.0071, "grad_norm": 1.0385583639144897, "learning_rate": 6.5181818181818195e-06, "num_tokens": 2592268.0, "completions/mean_length": 213.375, "completions/min_length": 212.0, "completions/max_length": 218.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 213.375, "completions/min_terminated_length": 212.0, "completions/max_terminated_length": 218.0, "rewards/meter/mean": 0.996587872505188, "rewards/meter/std": 0.0006334602949209511, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6363636255264282, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5073539018630981, "rewards/total_composite/std": 0.0003224806860089302, "reward": 0.5073539018630981, "reward_std": 0.00032248429488390684, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0020132821518927813, "sampling/sampling_logp_difference/max": 0.6653909683227539, "sampling/importance_sampling_ratio/min": 0.5140725374221802, "sampling/importance_sampling_ratio/mean": 1.000615119934082, "sampling/importance_sampling_ratio/max": 1.7144917249679565, "entropy": 0.006337960890959948, "clip_ratio/low_mean": 0.000589622650295496, "clip_ratio/low_min": 0.000589622650295496, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.000589622650295496, "reward_total_mean": 0.5073539018630981, "reward_meter_mean": 0.996587872505188, "reward_meter_std": 0.0006334602949209511, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6363636255264282, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5073539018630981, "reward_total_composite_std": 0.0003224806860089302} {"timestamp_utc": "2026-04-11T23:40:50Z", "mode": "eval", "global_step": 1150, "epoch": 0.04619030405269711, "eval_loss": NaN, "eval_runtime": 69.6474, "eval_samples_per_second": 1.493, "eval_steps_per_second": 0.187, "eval_num_tokens": 2592268.0, "eval_completions/mean_length": 209.30769230769232, "eval_completions/min_length": 63.53846153846154, "eval_completions/max_length": 363.6923076923077, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 209.30769230769232, "eval_completions/min_terminated_length": 63.53846153846154, "eval_completions/max_terminated_length": 363.6923076923077, "eval_rewards/meter/mean": 0.6754777202239404, "eval_rewards/meter/std": 0.40071778343274045, "eval_rewards/count_adherence/mean": 0.8692530210201557, "eval_rewards/count_adherence/std": 0.11666750907897949, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/repeat_penalty/mean": 0.6637609280072726, "eval_rewards/repeat_penalty/std": 0.22615024103568152, "eval_rewards/total_composite/mean": 0.39963403802651626, "eval_rewards/total_composite/std": 0.3056422472000122, "eval_reward": 0.39963403802651626, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.007615503783409412, "eval_sampling/sampling_logp_difference/max": 0.8880168061990005, "eval_sampling/importance_sampling_ratio/min": 0.4256615604345615, "eval_sampling/importance_sampling_ratio/mean": 1.001767112658574, "eval_sampling/importance_sampling_ratio/max": 1.3572269219618578, "eval_entropy": 0.06449572856609638, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.39963403802651626, "eval_reward_meter_mean": 0.6754777202239404, "eval_reward_meter_std": 0.40071778343274045, "eval_reward_count_adherence_mean": 0.8692530210201557, "eval_reward_count_adherence_std": 0.11666750907897949, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_repeat_penalty_mean": 0.6637609280072726, "eval_reward_repeat_penalty_std": 0.22615024103568152, "eval_reward_total_composite_mean": 0.39963403802651626, "eval_reward_total_composite_std": 0.3056422472000122} {"timestamp_utc": "2026-04-11T23:40:58Z", "mode": "train", "global_step": 1151, "epoch": 0.046230469534482066, "loss": -0.0045, "grad_norm": 0.24846500158309937, "learning_rate": 6.515151515151516e-06, "num_tokens": 2595104.0, "completions/mean_length": 180.5, "completions/min_length": 177.0, "completions/max_length": 181.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 180.5, "completions/min_terminated_length": 177.0, "completions/max_terminated_length": 181.0, "rewards/meter/mean": 0.9975081086158752, "rewards/meter/std": 0.000507785240188241, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4987540543079376, "rewards/total_composite/std": 0.0002538926200941205, "reward": 0.4987540543079376, "reward_std": 0.0002538926200941205, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0012201471254229546, "sampling/sampling_logp_difference/max": 0.5224804878234863, "sampling/importance_sampling_ratio/min": 0.90339595079422, "sampling/importance_sampling_ratio/mean": 1.0008914470672607, "sampling/importance_sampling_ratio/max": 1.6862051486968994, "entropy": 0.0079670658451505, "clip_ratio/low_mean": 0.0007062146905809641, "clip_ratio/low_min": 0.0007062146905809641, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0007062146905809641, "reward_total_mean": 0.4987540543079376, "reward_meter_mean": 0.9975081086158752, "reward_meter_std": 0.000507785240188241, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.4987540543079376, "reward_total_composite_std": 0.0002538926200941205} {"timestamp_utc": "2026-04-11T23:41:03Z", "mode": "train", "global_step": 1152, "epoch": 0.04627063501626702, "loss": 0.0463, "grad_norm": 17.063459396362305, "learning_rate": 6.512121212121213e-06, "num_tokens": 2596884.0, "completions/mean_length": 66.5, "completions/min_length": 57.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.5, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.8263880014419556, "rewards/meter/std": 0.2611541450023651, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8263880014419556, "rewards/total_composite/std": 0.2611541450023651, "reward": 0.8263880014419556, "reward_std": 0.2611541450023651, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06748796254396439, "sampling/sampling_logp_difference/max": 4.3918986320495605, "sampling/importance_sampling_ratio/min": 0.012377207167446613, "sampling/importance_sampling_ratio/mean": 1.0026739835739136, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3189387395977974, "clip_ratio/low_mean": 0.020571128465235233, "clip_ratio/low_min": 0.020571128465235233, "clip_ratio/high_mean": 0.03683905629441142, "clip_ratio/high_max": 0.03683905629441142, "clip_ratio/region_mean": 0.057410184759646654, "reward_total_mean": 0.8263880014419556, "reward_meter_mean": 0.8263880014419556, "reward_meter_std": 0.2611541450023651, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8263880014419556, "reward_total_composite_std": 0.2611541450023651} {"timestamp_utc": "2026-04-11T23:41:08Z", "mode": "train", "global_step": 1153, "epoch": 0.046310800498051974, "loss": 0.0129, "grad_norm": 3.4035446643829346, "learning_rate": 6.5090909090909095e-06, "num_tokens": 2599043.0, "completions/mean_length": 114.875, "completions/min_length": 112.0, "completions/max_length": 119.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.875, "completions/min_terminated_length": 112.0, "completions/max_terminated_length": 119.0, "rewards/meter/mean": 0.9962472319602966, "rewards/meter/std": 0.0029050002340227365, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8250000476837158, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8219736814498901, "rewards/total_composite/std": 0.07157647609710693, "reward": 0.8219736814498901, "reward_std": 0.07157646119594574, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0268792062997818, "sampling/sampling_logp_difference/max": 1.2601839303970337, "sampling/importance_sampling_ratio/min": 0.283601850271225, "sampling/importance_sampling_ratio/mean": 0.9999147057533264, "sampling/importance_sampling_ratio/max": 1.9995946884155273, "entropy": 0.1615551235154271, "clip_ratio/low_mean": 0.018382577574811876, "clip_ratio/low_min": 0.018382577574811876, "clip_ratio/high_mean": 0.0021929824724793434, "clip_ratio/high_max": 0.0021929824724793434, "clip_ratio/region_mean": 0.02057556004729122, "reward_total_mean": 0.8219736814498901, "reward_meter_mean": 0.9962472319602966, "reward_meter_std": 0.0029050002340227365, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8250000476837158, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.8219736814498901, "reward_total_composite_std": 0.07157647609710693} {"timestamp_utc": "2026-04-11T23:41:13Z", "mode": "train", "global_step": 1154, "epoch": 0.04635096597983693, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.506060606060607e-06, "num_tokens": 2600507.0, "completions/mean_length": 37.0, "completions/min_length": 37.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.0, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9963316321372986, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9963316321372986, "rewards/total_composite/std": 0.0, "reward": 0.9963316321372986, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0014280767645686865, "sampling/sampling_logp_difference/max": 0.04976058006286621, "sampling/importance_sampling_ratio/min": 0.9514572024345398, "sampling/importance_sampling_ratio/mean": 1.0005978345870972, "sampling/importance_sampling_ratio/max": 1.0367541313171387, "entropy": 0.014172183815389872, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9963316321372986, "reward_meter_mean": 0.9963316321372986, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9963316321372986, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-11T23:41:19Z", "mode": "train", "global_step": 1155, "epoch": 0.04639113146162188, "loss": -0.0352, "grad_norm": 1.2776048183441162, "learning_rate": 6.503030303030303e-06, "num_tokens": 2603752.0, "completions/mean_length": 222.625, "completions/min_length": 213.0, "completions/max_length": 241.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 222.625, "completions/min_terminated_length": 213.0, "completions/max_terminated_length": 241.0, "rewards/meter/mean": 0.9973673820495605, "rewards/meter/std": 0.000317126396112144, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.3440934121608734, "rewards/repeat_penalty/std": 0.21156011521816254, "rewards/total_composite/mean": 0.2860187888145447, "rewards/total_composite/std": 0.17588701844215393, "reward": 0.2860187888145447, "reward_std": 0.17588701844215393, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012430640868842602, "sampling/sampling_logp_difference/max": 2.1552317142486572, "sampling/importance_sampling_ratio/min": 0.11587633937597275, "sampling/importance_sampling_ratio/mean": 0.9986491203308105, "sampling/importance_sampling_ratio/max": 1.5816864967346191, "entropy": 0.04108358500525355, "clip_ratio/low_mean": 0.005830406676977873, "clip_ratio/low_min": 0.005830406676977873, "clip_ratio/high_mean": 0.009366041573230177, "clip_ratio/high_max": 0.009366041573230177, "clip_ratio/region_mean": 0.01519644825020805, "reward_total_mean": 0.2860187888145447, "reward_meter_mean": 0.9973673820495605, "reward_meter_std": 0.000317126396112144, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.3440934121608734, "reward_repeat_penalty_std": 0.21156011521816254, "reward_total_composite_mean": 0.2860187888145447, "reward_total_composite_std": 0.17588701844215393} {"timestamp_utc": "2026-04-11T23:41:24Z", "mode": "train", "global_step": 1156, "epoch": 0.046431296943406836, "loss": 0.0178, "grad_norm": 7.644164562225342, "learning_rate": 6.5000000000000004e-06, "num_tokens": 2605292.0, "completions/mean_length": 33.5, "completions/min_length": 32.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.5, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9726088643074036, "rewards/meter/std": 0.043839771300554276, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9726088643074036, "rewards/total_composite/std": 0.043839771300554276, "reward": 0.9726088643074036, "reward_std": 0.04383978247642517, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.041076283901929855, "sampling/sampling_logp_difference/max": 1.6910080909729004, "sampling/importance_sampling_ratio/min": 0.18433360755443573, "sampling/importance_sampling_ratio/mean": 1.0058238506317139, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.16383972112089396, "clip_ratio/low_mean": 0.0036764706019312143, "clip_ratio/low_min": 0.0036764706019312143, "clip_ratio/high_mean": 0.026069519110023975, "clip_ratio/high_max": 0.026069519110023975, "clip_ratio/region_mean": 0.02974598971195519, "reward_total_mean": 0.9726088643074036, "reward_meter_mean": 0.9726088643074036, "reward_meter_std": 0.043839771300554276, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9726088643074036, "reward_total_composite_std": 0.043839771300554276} {"timestamp_utc": "2026-04-11T23:41:28Z", "mode": "train", "global_step": 1157, "epoch": 0.04647146242519179, "loss": 0.0036, "grad_norm": 5.127370357513428, "learning_rate": 6.496969696969697e-06, "num_tokens": 2607186.0, "completions/mean_length": 62.75, "completions/min_length": 61.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.75, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.4182388186454773, "rewards/meter/std": 0.25260862708091736, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4182388186454773, "rewards/total_composite/std": 0.25260862708091736, "reward": 0.4182388186454773, "reward_std": 0.25260859727859497, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03417876362800598, "sampling/sampling_logp_difference/max": 1.50307297706604, "sampling/importance_sampling_ratio/min": 0.2224455326795578, "sampling/importance_sampling_ratio/mean": 1.0079007148742676, "sampling/importance_sampling_ratio/max": 1.7306538820266724, "entropy": 0.1482592076063156, "clip_ratio/low_mean": 0.03776956582441926, "clip_ratio/low_min": 0.03776956582441926, "clip_ratio/high_mean": 0.004000256070867181, "clip_ratio/high_max": 0.004000256070867181, "clip_ratio/region_mean": 0.04176982189528644, "reward_total_mean": 0.4182388186454773, "reward_meter_mean": 0.4182388186454773, "reward_meter_std": 0.25260862708091736, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.4182388186454773, "reward_total_composite_std": 0.25260862708091736} {"timestamp_utc": "2026-04-11T23:41:33Z", "mode": "train", "global_step": 1158, "epoch": 0.046511627906976744, "loss": 0.0042, "grad_norm": 3.4404122829437256, "learning_rate": 6.493939393939395e-06, "num_tokens": 2608844.0, "completions/mean_length": 55.25, "completions/min_length": 54.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.25, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.1349601447582245, "rewards/meter/std": 0.012392125092446804, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7083333730697632, "rewards/repeat_penalty/std": 0.11785111576318741, "rewards/total_composite/mean": 0.09517530351877213, "rewards/total_composite/std": 0.01430203951895237, "reward": 0.09517530351877213, "reward_std": 0.01430203951895237, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016672344878315926, "sampling/sampling_logp_difference/max": 1.2043662071228027, "sampling/importance_sampling_ratio/min": 0.2998819947242737, "sampling/importance_sampling_ratio/mean": 1.0016802549362183, "sampling/importance_sampling_ratio/max": 1.3683043718338013, "entropy": 0.07266200426965952, "clip_ratio/low_mean": 0.008780332282185555, "clip_ratio/low_min": 0.008780332282185555, "clip_ratio/high_mean": 0.004545454401522875, "clip_ratio/high_max": 0.004545454401522875, "clip_ratio/region_mean": 0.01332578668370843, "reward_total_mean": 0.09517530351877213, "reward_meter_mean": 0.1349601447582245, "reward_meter_std": 0.012392125092446804, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7083333730697632, "reward_repeat_penalty_std": 0.11785111576318741, "reward_total_composite_mean": 0.09517530351877213, "reward_total_composite_std": 0.01430203951895237} {"timestamp_utc": "2026-04-11T23:41:37Z", "mode": "train", "global_step": 1159, "epoch": 0.0465517933887617, "loss": 0.0034, "grad_norm": 10.866904258728027, "learning_rate": 6.490909090909091e-06, "num_tokens": 2610338.0, "completions/mean_length": 32.75, "completions/min_length": 32.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.75, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9911776781082153, "rewards/meter/std": 0.0027705347165465355, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9911776781082153, "rewards/total_composite/std": 0.0027705347165465355, "reward": 0.9911776781082153, "reward_std": 0.002770546358078718, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017460720613598824, "sampling/sampling_logp_difference/max": 0.969951868057251, "sampling/importance_sampling_ratio/min": 0.3791012763977051, "sampling/importance_sampling_ratio/mean": 1.002776861190796, "sampling/importance_sampling_ratio/max": 1.708295226097107, "entropy": 0.0744382287375629, "clip_ratio/low_mean": 0.011482007801532745, "clip_ratio/low_min": 0.011482007801532745, "clip_ratio/high_mean": 0.0037878789007663727, "clip_ratio/high_max": 0.0037878789007663727, "clip_ratio/region_mean": 0.015269886702299118, "reward_total_mean": 0.9911776781082153, "reward_meter_mean": 0.9911776781082153, "reward_meter_std": 0.0027705347165465355, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9911776781082153, "reward_total_composite_std": 0.0027705347165465355} {"timestamp_utc": "2026-04-11T23:41:41Z", "mode": "train", "global_step": 1160, "epoch": 0.04659195887054665, "loss": 0.0007, "grad_norm": 5.433602809906006, "learning_rate": 6.487878787878789e-06, "num_tokens": 2612174.0, "completions/mean_length": 56.5, "completions/min_length": 53.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.5, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.19516333937644958, "rewards/meter/std": 0.10145802795886993, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.19516333937644958, "rewards/total_composite/std": 0.10145802795886993, "reward": 0.19516333937644958, "reward_std": 0.10145802050828934, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030529655516147614, "sampling/sampling_logp_difference/max": 4.638236999511719, "sampling/importance_sampling_ratio/min": 0.009674739092588425, "sampling/importance_sampling_ratio/mean": 1.0025193691253662, "sampling/importance_sampling_ratio/max": 1.9637987613677979, "entropy": 0.07584757776930928, "clip_ratio/low_mean": 0.004551473073661327, "clip_ratio/low_min": 0.004551473073661327, "clip_ratio/high_mean": 0.0021929824724793434, "clip_ratio/high_max": 0.0021929824724793434, "clip_ratio/region_mean": 0.006744455546140671, "reward_total_mean": 0.19516333937644958, "reward_meter_mean": 0.19516333937644958, "reward_meter_std": 0.10145802795886993, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.19516333937644958, "reward_total_composite_std": 0.10145802795886993} {"timestamp_utc": "2026-04-11T23:41:46Z", "mode": "train", "global_step": 1161, "epoch": 0.046632124352331605, "loss": -0.0148, "grad_norm": 12.443059921264648, "learning_rate": 6.484848484848485e-06, "num_tokens": 2613517.0, "completions/mean_length": 31.875, "completions/min_length": 31.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.875, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9917822480201721, "rewards/meter/std": 0.003353093983605504, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9917822480201721, "rewards/total_composite/std": 0.003353093983605504, "reward": 0.9917822480201721, "reward_std": 0.003353102831169963, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02346481755375862, "sampling/sampling_logp_difference/max": 0.5687723755836487, "sampling/importance_sampling_ratio/min": 0.6050869226455688, "sampling/importance_sampling_ratio/mean": 1.0028786659240723, "sampling/importance_sampling_ratio/max": 1.7660976648330688, "entropy": 0.11351501289755106, "clip_ratio/low_mean": 0.02016128972172737, "clip_ratio/low_min": 0.02016128972172737, "clip_ratio/high_mean": 0.011600378900766373, "clip_ratio/high_max": 0.011600378900766373, "clip_ratio/region_mean": 0.031761668622493744, "reward_total_mean": 0.9917822480201721, "reward_meter_mean": 0.9917822480201721, "reward_meter_std": 0.003353093983605504, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9917822480201721, "reward_total_composite_std": 0.003353093983605504} {"timestamp_utc": "2026-04-11T23:41:51Z", "mode": "train", "global_step": 1162, "epoch": 0.04667228983411656, "loss": 0.0034, "grad_norm": 3.7559328079223633, "learning_rate": 6.481818181818182e-06, "num_tokens": 2615967.0, "completions/mean_length": 116.25, "completions/min_length": 114.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.25, "completions/min_terminated_length": 114.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.9908610582351685, "rewards/meter/std": 0.010183922946453094, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.891869068145752, "rewards/total_composite/std": 0.10717974603176117, "reward": 0.891869068145752, "reward_std": 0.10717973858118057, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03219734504818916, "sampling/sampling_logp_difference/max": 1.7524470090866089, "sampling/importance_sampling_ratio/min": 0.17334923148155212, "sampling/importance_sampling_ratio/mean": 0.9961613416671753, "sampling/importance_sampling_ratio/max": 1.661534309387207, "entropy": 0.16215351317077875, "clip_ratio/low_mean": 0.005444317124783993, "clip_ratio/low_min": 0.005444317124783993, "clip_ratio/high_mean": 0.027328122640028596, "clip_ratio/high_max": 0.027328122640028596, "clip_ratio/region_mean": 0.03277243976481259, "reward_total_mean": 0.891869068145752, "reward_meter_mean": 0.9908610582351685, "reward_meter_std": 0.010183922946453094, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_total_composite_mean": 0.891869068145752, "reward_total_composite_std": 0.10717974603176117} {"timestamp_utc": "2026-04-11T23:41:57Z", "mode": "train", "global_step": 1163, "epoch": 0.04671245531590151, "loss": -0.0674, "grad_norm": 2.036916971206665, "learning_rate": 6.478787878787879e-06, "num_tokens": 2619154.0, "completions/mean_length": 211.375, "completions/min_length": 152.0, "completions/max_length": 232.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 211.375, "completions/min_terminated_length": 152.0, "completions/max_terminated_length": 232.0, "rewards/meter/mean": 0.8547403812408447, "rewards/meter/std": 0.34573426842689514, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4711538553237915, "rewards/repeat_penalty/std": 0.2268439084291458, "rewards/total_composite/mean": 0.31233876943588257, "rewards/total_composite/std": 0.20915131270885468, "reward": 0.31233876943588257, "reward_std": 0.20915129780769348, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010163981467485428, "sampling/sampling_logp_difference/max": 0.8715519905090332, "sampling/importance_sampling_ratio/min": 0.41830188035964966, "sampling/importance_sampling_ratio/mean": 0.9999427795410156, "sampling/importance_sampling_ratio/max": 1.551490068435669, "entropy": 0.044531215680763125, "clip_ratio/low_mean": 0.003117707441560924, "clip_ratio/low_min": 0.003117707441560924, "clip_ratio/high_mean": 0.003341847565025091, "clip_ratio/high_max": 0.003341847565025091, "clip_ratio/region_mean": 0.006459555006586015, "reward_total_mean": 0.31233876943588257, "reward_meter_mean": 0.8547403812408447, "reward_meter_std": 0.34573426842689514, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4711538553237915, "reward_repeat_penalty_std": 0.2268439084291458, "reward_total_composite_mean": 0.31233876943588257, "reward_total_composite_std": 0.20915131270885468} {"timestamp_utc": "2026-04-11T23:42:02Z", "mode": "train", "global_step": 1164, "epoch": 0.04675262079768647, "loss": 0.0062, "grad_norm": 4.857386112213135, "learning_rate": 6.475757575757576e-06, "num_tokens": 2621022.0, "completions/mean_length": 75.5, "completions/min_length": 75.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.5, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.905701756477356, "rewards/meter/std": 0.22805918753147125, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.905701756477356, "rewards/total_composite/std": 0.22805918753147125, "reward": 0.905701756477356, "reward_std": 0.22805917263031006, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014857825823128223, "sampling/sampling_logp_difference/max": 0.9470729827880859, "sampling/importance_sampling_ratio/min": 0.38787469267845154, "sampling/importance_sampling_ratio/mean": 1.003965139389038, "sampling/importance_sampling_ratio/max": 1.5772894620895386, "entropy": 0.06199400080367923, "clip_ratio/low_mean": 0.0016447368543595076, "clip_ratio/low_min": 0.0016447368543595076, "clip_ratio/high_mean": 0.014912280952557921, "clip_ratio/high_max": 0.014912280952557921, "clip_ratio/region_mean": 0.01655701780691743, "reward_total_mean": 0.905701756477356, "reward_meter_mean": 0.905701756477356, "reward_meter_std": 0.22805918753147125, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.905701756477356, "reward_total_composite_std": 0.22805918753147125} {"timestamp_utc": "2026-04-11T23:42:06Z", "mode": "train", "global_step": 1165, "epoch": 0.04679278627947142, "loss": 0.0288, "grad_norm": 4.955336570739746, "learning_rate": 6.472727272727272e-06, "num_tokens": 2622496.0, "completions/mean_length": 40.25, "completions/min_length": 37.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.25, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9979323148727417, "rewards/meter/std": 0.0005710619152523577, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979323148727417, "rewards/total_composite/std": 0.0005710619152523577, "reward": 0.9979323148727417, "reward_std": 0.0005710580153390765, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05128004774451256, "sampling/sampling_logp_difference/max": 1.0725364685058594, "sampling/importance_sampling_ratio/min": 0.3421395719051361, "sampling/importance_sampling_ratio/mean": 0.9936442971229553, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.19564981944859028, "clip_ratio/low_mean": 0.009149970952421427, "clip_ratio/low_min": 0.009149970952421427, "clip_ratio/high_mean": 0.02869208576157689, "clip_ratio/high_max": 0.02869208576157689, "clip_ratio/region_mean": 0.03784205671399832, "reward_total_mean": 0.9979323148727417, "reward_meter_mean": 0.9979323148727417, "reward_meter_std": 0.0005710619152523577, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9979323148727417, "reward_total_composite_std": 0.0005710619152523577} {"timestamp_utc": "2026-04-11T23:42:11Z", "mode": "train", "global_step": 1166, "epoch": 0.046832951761256375, "loss": -0.0054, "grad_norm": 10.79326057434082, "learning_rate": 6.4696969696969705e-06, "num_tokens": 2624300.0, "completions/mean_length": 66.5, "completions/min_length": 63.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.5, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9930996894836426, "rewards/meter/std": 0.0033798664808273315, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9930996894836426, "rewards/total_composite/std": 0.0033798664808273315, "reward": 0.9930996894836426, "reward_std": 0.003379874862730503, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02763393148779869, "sampling/sampling_logp_difference/max": 2.4080264568328857, "sampling/importance_sampling_ratio/min": 0.08999272435903549, "sampling/importance_sampling_ratio/mean": 0.9978261590003967, "sampling/importance_sampling_ratio/max": 1.7809786796569824, "entropy": 0.0714933155104518, "clip_ratio/low_mean": 0.007695082924328744, "clip_ratio/low_min": 0.007695082924328744, "clip_ratio/high_mean": 0.014788191299885511, "clip_ratio/high_max": 0.014788191299885511, "clip_ratio/region_mean": 0.022483274224214256, "reward_total_mean": 0.9930996894836426, "reward_meter_mean": 0.9930996894836426, "reward_meter_std": 0.0033798664808273315, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9930996894836426, "reward_total_composite_std": 0.0033798664808273315} {"timestamp_utc": "2026-04-11T23:42:17Z", "mode": "train", "global_step": 1167, "epoch": 0.04687311724304133, "loss": 0.0106, "grad_norm": 0.7076108455657959, "learning_rate": 6.466666666666667e-06, "num_tokens": 2627412.0, "completions/mean_length": 203.0, "completions/min_length": 197.0, "completions/max_length": 210.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 203.0, "completions/min_terminated_length": 197.0, "completions/max_terminated_length": 210.0, "rewards/meter/mean": 0.07068999111652374, "rewards/meter/std": 0.12723270058631897, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6153846383094788, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.03625128045678139, "rewards/total_composite/std": 0.0652475357055664, "reward": 0.03625128045678139, "reward_std": 0.06524752825498581, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005202981643378735, "sampling/sampling_logp_difference/max": 2.0959839820861816, "sampling/importance_sampling_ratio/min": 0.12294920533895493, "sampling/importance_sampling_ratio/mean": 1.00026535987854, "sampling/importance_sampling_ratio/max": 1.9165208339691162, "entropy": 0.020221689250320196, "clip_ratio/low_mean": 0.005486392183229327, "clip_ratio/low_min": 0.005486392183229327, "clip_ratio/high_mean": 0.0012690355069935322, "clip_ratio/high_max": 0.0012690355069935322, "clip_ratio/region_mean": 0.006755427690222859, "reward_total_mean": 0.03625128045678139, "reward_meter_mean": 0.07068999111652374, "reward_meter_std": 0.12723270058631897, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6153846383094788, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.03625128045678139, "reward_total_composite_std": 0.0652475357055664} {"timestamp_utc": "2026-04-11T23:42:21Z", "mode": "train", "global_step": 1168, "epoch": 0.04691328272482628, "loss": 0.0086, "grad_norm": 3.631923198699951, "learning_rate": 6.463636363636364e-06, "num_tokens": 2629391.0, "completions/mean_length": 86.375, "completions/min_length": 85.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.375, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.07566258311271667, "rewards/meter/std": 0.10104013234376907, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.06269732117652893, "rewards/total_composite/std": 0.08014771342277527, "reward": 0.06269732117652893, "reward_std": 0.08014770597219467, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009703797288239002, "sampling/sampling_logp_difference/max": 1.0345752239227295, "sampling/importance_sampling_ratio/min": 0.35537731647491455, "sampling/importance_sampling_ratio/mean": 0.9974919557571411, "sampling/importance_sampling_ratio/max": 1.388146162033081, "entropy": 0.04172563157044351, "clip_ratio/low_mean": 0.0043441514717414975, "clip_ratio/low_min": 0.0043441514717414975, "clip_ratio/high_mean": 0.00147058826405555, "clip_ratio/high_max": 0.00147058826405555, "clip_ratio/region_mean": 0.005814739735797048, "reward_total_mean": 0.06269732117652893, "reward_meter_mean": 0.07566258311271667, "reward_meter_std": 0.10104013234376907, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.06269732117652893, "reward_total_composite_std": 0.08014771342277527} {"timestamp_utc": "2026-04-11T23:42:29Z", "mode": "train", "global_step": 1169, "epoch": 0.04695344820661124, "loss": -0.0418, "grad_norm": 1.607576847076416, "learning_rate": 6.460606060606061e-06, "num_tokens": 2633025.0, "completions/mean_length": 262.25, "completions/min_length": 234.0, "completions/max_length": 285.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 262.25, "completions/min_terminated_length": 234.0, "completions/max_terminated_length": 285.0, "rewards/meter/mean": 0.9962061643600464, "rewards/meter/std": 0.0021561330650001764, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6634615659713745, "rewards/repeat_penalty/std": 0.05723259598016739, "rewards/total_composite/mean": 0.5507833957672119, "rewards/total_composite/std": 0.04750329256057739, "reward": 0.5507833957672119, "reward_std": 0.04750329256057739, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010979145765304565, "sampling/sampling_logp_difference/max": 1.3150863647460938, "sampling/importance_sampling_ratio/min": 0.26845112442970276, "sampling/importance_sampling_ratio/mean": 1.0013923645019531, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.047941830940544605, "clip_ratio/low_mean": 0.0029717457364313304, "clip_ratio/low_min": 0.0029717457364313304, "clip_ratio/high_mean": 0.00317868051934056, "clip_ratio/high_max": 0.00317868051934056, "clip_ratio/region_mean": 0.00615042625577189, "reward_total_mean": 0.5507833957672119, "reward_meter_mean": 0.9962061643600464, "reward_meter_std": 0.0021561330650001764, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6634615659713745, "reward_repeat_penalty_std": 0.05723259598016739, "reward_total_composite_mean": 0.5507833957672119, "reward_total_composite_std": 0.04750329256057739} {"timestamp_utc": "2026-04-11T23:42:33Z", "mode": "train", "global_step": 1170, "epoch": 0.04699361368839619, "loss": -0.0059, "grad_norm": 4.66847562789917, "learning_rate": 6.457575757575758e-06, "num_tokens": 2635041.0, "completions/mean_length": 80.0, "completions/min_length": 79.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.0, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.9650785326957703, "rewards/meter/std": 0.06653696298599243, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9650785326957703, "rewards/total_composite/std": 0.06653696298599243, "reward": 0.9650785326957703, "reward_std": 0.06653697043657303, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03951912745833397, "sampling/sampling_logp_difference/max": 1.5156645774841309, "sampling/importance_sampling_ratio/min": 0.2196621596813202, "sampling/importance_sampling_ratio/mean": 1.0022578239440918, "sampling/importance_sampling_ratio/max": 1.8339917659759521, "entropy": 0.24884118884801865, "clip_ratio/low_mean": 0.0031645570416003466, "clip_ratio/low_min": 0.0031645570416003466, "clip_ratio/high_mean": 0.02818627143278718, "clip_ratio/high_max": 0.02818627143278718, "clip_ratio/region_mean": 0.031350828474387527, "reward_total_mean": 0.9650785326957703, "reward_meter_mean": 0.9650785326957703, "reward_meter_std": 0.06653696298599243, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9650785326957703, "reward_total_composite_std": 0.06653696298599243} {"timestamp_utc": "2026-04-11T23:42:38Z", "mode": "train", "global_step": 1171, "epoch": 0.047033779170181145, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.454545454545456e-06, "num_tokens": 2636953.0, "completions/mean_length": 85.0, "completions/min_length": 85.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.0, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.3254895508289337, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.2603916525840759, "rewards/total_composite/std": 0.0, "reward": 0.2603916525840759, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0026430224534124136, "sampling/sampling_logp_difference/max": 0.10097821056842804, "sampling/importance_sampling_ratio/min": 0.9174838662147522, "sampling/importance_sampling_ratio/mean": 1.0015860795974731, "sampling/importance_sampling_ratio/max": 1.1062525510787964, "entropy": 0.023862487636506557, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.2603916525840759, "reward_meter_mean": 0.3254895508289337, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.2603916525840759, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-11T23:42:43Z", "mode": "train", "global_step": 1172, "epoch": 0.0470739446519661, "loss": 0.008, "grad_norm": 4.03551721572876, "learning_rate": 6.451515151515152e-06, "num_tokens": 2639474.0, "completions/mean_length": 118.125, "completions/min_length": 114.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.125, "completions/min_terminated_length": 114.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.9958237409591675, "rewards/meter/std": 0.0014982123393565416, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9249999523162842, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.921150803565979, "rewards/total_composite/std": 0.10321994870901108, "reward": 0.921150803565979, "reward_std": 0.10321993380784988, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027086030691862106, "sampling/sampling_logp_difference/max": 1.4236351251602173, "sampling/importance_sampling_ratio/min": 0.2408369481563568, "sampling/importance_sampling_ratio/mean": 1.001880407333374, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.13357019051909447, "clip_ratio/low_mean": 0.010513697401620448, "clip_ratio/low_min": 0.010513697401620448, "clip_ratio/high_mean": 0.015921411802992225, "clip_ratio/high_max": 0.015921411802992225, "clip_ratio/region_mean": 0.026435109204612672, "reward_total_mean": 0.921150803565979, "reward_meter_mean": 0.9958237409591675, "reward_meter_std": 0.0014982123393565416, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9249999523162842, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.921150803565979, "reward_total_composite_std": 0.10321994870901108} {"timestamp_utc": "2026-04-11T23:42:48Z", "mode": "train", "global_step": 1173, "epoch": 0.04711411013375105, "loss": 0.008, "grad_norm": 3.1960055828094482, "learning_rate": 6.4484848484848496e-06, "num_tokens": 2641397.0, "completions/mean_length": 76.375, "completions/min_length": 74.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.375, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.988898754119873, "rewards/meter/std": 0.0167799461632967, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.988898754119873, "rewards/total_composite/std": 0.0167799461632967, "reward": 0.988898754119873, "reward_std": 0.016779951751232147, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015661096200346947, "sampling/sampling_logp_difference/max": 1.2024707794189453, "sampling/importance_sampling_ratio/min": 0.3004509508609772, "sampling/importance_sampling_ratio/mean": 1.0082311630249023, "sampling/importance_sampling_ratio/max": 1.5719304084777832, "entropy": 0.08165451744571328, "clip_ratio/low_mean": 0.004870129749178886, "clip_ratio/low_min": 0.004870129749178886, "clip_ratio/high_mean": 0.008183791185729206, "clip_ratio/high_max": 0.008183791185729206, "clip_ratio/region_mean": 0.013053920934908092, "reward_total_mean": 0.988898754119873, "reward_meter_mean": 0.988898754119873, "reward_meter_std": 0.0167799461632967, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.988898754119873, "reward_total_composite_std": 0.0167799461632967} {"timestamp_utc": "2026-04-11T23:42:52Z", "mode": "train", "global_step": 1174, "epoch": 0.04715427561553601, "loss": -0.0038, "grad_norm": 3.3341424465179443, "learning_rate": 6.445454545454546e-06, "num_tokens": 2643116.0, "completions/mean_length": 75.875, "completions/min_length": 75.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.875, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9900845289230347, "rewards/meter/std": 0.006058650091290474, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9900845289230347, "rewards/total_composite/std": 0.006058650091290474, "reward": 0.9900845289230347, "reward_std": 0.006058644037693739, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017089035362005234, "sampling/sampling_logp_difference/max": 0.9457058906555176, "sampling/importance_sampling_ratio/min": 0.38840532302856445, "sampling/importance_sampling_ratio/mean": 1.0023661851882935, "sampling/importance_sampling_ratio/max": 1.7328146696090698, "entropy": 0.08360559307038784, "clip_ratio/low_mean": 0.0033333334140479565, "clip_ratio/low_min": 0.0033333334140479565, "clip_ratio/high_mean": 0.003289473708719015, "clip_ratio/high_max": 0.003289473708719015, "clip_ratio/region_mean": 0.006622807122766972, "reward_total_mean": 0.9900845289230347, "reward_meter_mean": 0.9900845289230347, "reward_meter_std": 0.006058650091290474, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9900845289230347, "reward_total_composite_std": 0.006058650091290474} {"timestamp_utc": "2026-04-11T23:42:57Z", "mode": "train", "global_step": 1175, "epoch": 0.04719444109732096, "loss": 0.0211, "grad_norm": 3.098088502883911, "learning_rate": 6.442424242424243e-06, "num_tokens": 2645197.0, "completions/mean_length": 86.125, "completions/min_length": 85.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.125, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.2395191788673401, "rewards/meter/std": 0.14119157195091248, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.19161534309387207, "rewards/total_composite/std": 0.11295326054096222, "reward": 0.19161534309387207, "reward_std": 0.11295326054096222, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008148287422955036, "sampling/sampling_logp_difference/max": 1.1111087799072266, "sampling/importance_sampling_ratio/min": 0.32919374108314514, "sampling/importance_sampling_ratio/mean": 1.0021647214889526, "sampling/importance_sampling_ratio/max": 1.5297051668167114, "entropy": 0.05733899096958339, "clip_ratio/low_mean": 0.002794080995954573, "clip_ratio/low_min": 0.002794080995954573, "clip_ratio/high_mean": 0.004411764908581972, "clip_ratio/high_max": 0.004411764908581972, "clip_ratio/region_mean": 0.007205845904536545, "reward_total_mean": 0.19161534309387207, "reward_meter_mean": 0.2395191788673401, "reward_meter_std": 0.14119157195091248, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.19161534309387207, "reward_total_composite_std": 0.11295326054096222} {"timestamp_utc": "2026-04-11T23:43:05Z", "mode": "train", "global_step": 1176, "epoch": 0.047234606579105914, "loss": -0.0027, "grad_norm": 1.6563795804977417, "learning_rate": 6.43939393939394e-06, "num_tokens": 2649214.0, "completions/mean_length": 277.125, "completions/min_length": 268.0, "completions/max_length": 292.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 277.125, "completions/min_terminated_length": 268.0, "completions/max_terminated_length": 292.0, "rewards/meter/mean": 0.9914854764938354, "rewards/meter/std": 0.0076880063861608505, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7403846383094788, "rewards/repeat_penalty/std": 0.15350237488746643, "rewards/total_composite/mean": 0.6114773750305176, "rewards/total_composite/std": 0.12588950991630554, "reward": 0.6114773750305176, "reward_std": 0.12588950991630554, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015588824637234211, "sampling/sampling_logp_difference/max": 1.1217985153198242, "sampling/importance_sampling_ratio/min": 0.3256934881210327, "sampling/importance_sampling_ratio/mean": 1.0004220008850098, "sampling/importance_sampling_ratio/max": 1.965925931930542, "entropy": 0.09004434384405613, "clip_ratio/low_mean": 0.0026997023087460548, "clip_ratio/low_min": 0.0026997023087460548, "clip_ratio/high_mean": 0.013525328016839921, "clip_ratio/high_max": 0.013525328016839921, "clip_ratio/region_mean": 0.016225030325585976, "reward_total_mean": 0.6114773750305176, "reward_meter_mean": 0.9914854764938354, "reward_meter_std": 0.0076880063861608505, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7403846383094788, "reward_repeat_penalty_std": 0.15350237488746643, "reward_total_composite_mean": 0.6114773750305176, "reward_total_composite_std": 0.12588950991630554} {"timestamp_utc": "2026-04-11T23:43:10Z", "mode": "train", "global_step": 1177, "epoch": 0.04727477206089087, "loss": -0.0359, "grad_norm": 4.175478935241699, "learning_rate": 6.436363636363637e-06, "num_tokens": 2650978.0, "completions/mean_length": 58.5, "completions/min_length": 57.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.5, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.36279141902923584, "rewards/meter/std": 0.24407321214675903, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.36279141902923584, "rewards/total_composite/std": 0.24407321214675903, "reward": 0.36279141902923584, "reward_std": 0.24407321214675903, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009452205151319504, "sampling/sampling_logp_difference/max": 0.6868042945861816, "sampling/importance_sampling_ratio/min": 0.503181517124176, "sampling/importance_sampling_ratio/mean": 1.0017210245132446, "sampling/importance_sampling_ratio/max": 1.3822486400604248, "entropy": 0.06206447468139231, "clip_ratio/low_mean": 0.006048386916518211, "clip_ratio/low_min": 0.006048386916518211, "clip_ratio/high_mean": 0.00390625, "clip_ratio/high_max": 0.00390625, "clip_ratio/region_mean": 0.009954636916518211, "reward_total_mean": 0.36279141902923584, "reward_meter_mean": 0.36279141902923584, "reward_meter_std": 0.24407321214675903, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.36279141902923584, "reward_total_composite_std": 0.24407321214675903} {"timestamp_utc": "2026-04-11T23:43:19Z", "mode": "train", "global_step": 1178, "epoch": 0.04731493754267582, "loss": -0.1566, "grad_norm": 1.462255835533142, "learning_rate": 6.433333333333333e-06, "num_tokens": 2653979.0, "completions/mean_length": 267.125, "completions/min_length": 218.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 232.1428680419922, "completions/min_terminated_length": 218.0, "completions/max_terminated_length": 257.0, "rewards/meter/mean": 0.4442686140537262, "rewards/meter/std": 0.2729472219944, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.05050762742757797, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.6192708611488342, "rewards/repeat_penalty/std": 0.21683759987354279, "rewards/total_composite/mean": 0.2283807098865509, "rewards/total_composite/std": 0.17121197283267975, "reward": 0.2283807098865509, "reward_std": 0.17121195793151855, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012943691574037075, "sampling/sampling_logp_difference/max": 1.0156912803649902, "sampling/importance_sampling_ratio/min": 0.44583451747894287, "sampling/importance_sampling_ratio/mean": 1.0050114393234253, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07222116971388459, "clip_ratio/low_mean": 0.0021399176912382245, "clip_ratio/low_min": 0.0021399176912382245, "clip_ratio/high_mean": 0.007645888428669423, "clip_ratio/high_max": 0.007645888428669423, "clip_ratio/region_mean": 0.009785806119907647, "reward_total_mean": 0.2283807098865509, "reward_meter_mean": 0.4442686140537262, "reward_meter_std": 0.2729472219944, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.05050762742757797, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.6192708611488342, "reward_repeat_penalty_std": 0.21683759987354279, "reward_total_composite_mean": 0.2283807098865509, "reward_total_composite_std": 0.17121197283267975} {"timestamp_utc": "2026-04-11T23:43:24Z", "mode": "train", "global_step": 1179, "epoch": 0.047355103024460776, "loss": 0.0133, "grad_norm": 11.408062934875488, "learning_rate": 6.430303030303031e-06, "num_tokens": 2655696.0, "completions/mean_length": 68.625, "completions/min_length": 66.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.625, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9946950078010559, "rewards/meter/std": 0.002590329386293888, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9946950078010559, "rewards/total_composite/std": 0.002590329386293888, "reward": 0.9946950078010559, "reward_std": 0.002590324031189084, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026325104758143425, "sampling/sampling_logp_difference/max": 1.1628682613372803, "sampling/importance_sampling_ratio/min": 0.31258830428123474, "sampling/importance_sampling_ratio/mean": 1.0001428127288818, "sampling/importance_sampling_ratio/max": 1.6397497653961182, "entropy": 0.09837264008820057, "clip_ratio/low_mean": 0.014470615307800472, "clip_ratio/low_min": 0.014470615307800472, "clip_ratio/high_mean": 0.016267166705802083, "clip_ratio/high_max": 0.016267166705802083, "clip_ratio/region_mean": 0.030737782013602555, "reward_total_mean": 0.9946950078010559, "reward_meter_mean": 0.9946950078010559, "reward_meter_std": 0.002590329386293888, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9946950078010559, "reward_total_composite_std": 0.002590329386293888} {"timestamp_utc": "2026-04-11T23:43:28Z", "mode": "train", "global_step": 1180, "epoch": 0.04739526850624573, "loss": 0.0028, "grad_norm": 7.856446266174316, "learning_rate": 6.427272727272728e-06, "num_tokens": 2657383.0, "completions/mean_length": 55.875, "completions/min_length": 52.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.875, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.7185721397399902, "rewards/meter/std": 0.25754180550575256, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7185721397399902, "rewards/total_composite/std": 0.25754180550575256, "reward": 0.7185721397399902, "reward_std": 0.25754180550575256, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03376821056008339, "sampling/sampling_logp_difference/max": 1.3033967018127441, "sampling/importance_sampling_ratio/min": 0.2716076672077179, "sampling/importance_sampling_ratio/mean": 0.9931161403656006, "sampling/importance_sampling_ratio/max": 1.7777073383331299, "entropy": 0.142734014429152, "clip_ratio/low_mean": 0.011504121124744415, "clip_ratio/low_min": 0.011504121124744415, "clip_ratio/high_mean": 0.0242516309954226, "clip_ratio/high_max": 0.0242516309954226, "clip_ratio/region_mean": 0.03575575212016702, "reward_total_mean": 0.7185721397399902, "reward_meter_mean": 0.7185721397399902, "reward_meter_std": 0.25754180550575256, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7185721397399902, "reward_total_composite_std": 0.25754180550575256} {"timestamp_utc": "2026-04-11T23:43:37Z", "mode": "train", "global_step": 1181, "epoch": 0.047435433988030684, "loss": 0.0075, "grad_norm": 1.3380672931671143, "learning_rate": 6.424242424242425e-06, "num_tokens": 2662002.0, "completions/mean_length": 371.375, "completions/min_length": 359.0, "completions/max_length": 384.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 371.375, "completions/min_terminated_length": 359.0, "completions/max_terminated_length": 384.0, "rewards/meter/mean": 0.9918642044067383, "rewards/meter/std": 0.006125128827989101, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7426470518112183, "rewards/repeat_penalty/std": 0.08281681686639786, "rewards/total_composite/mean": 0.6446620225906372, "rewards/total_composite/std": 0.07332957535982132, "reward": 0.6446620225906372, "reward_std": 0.07332959026098251, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013688583858311176, "sampling/sampling_logp_difference/max": 1.4129068851470947, "sampling/importance_sampling_ratio/min": 0.2434346228837967, "sampling/importance_sampling_ratio/mean": 1.0013055801391602, "sampling/importance_sampling_ratio/max": 1.6264017820358276, "entropy": 0.0724606835283339, "clip_ratio/low_mean": 0.004685206571593881, "clip_ratio/low_min": 0.004685206571593881, "clip_ratio/high_mean": 0.006440639495849609, "clip_ratio/high_max": 0.006440639495849609, "clip_ratio/region_mean": 0.01112584606744349, "reward_total_mean": 0.6446620225906372, "reward_meter_mean": 0.9918642044067383, "reward_meter_std": 0.006125128827989101, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7426470518112183, "reward_repeat_penalty_std": 0.08281681686639786, "reward_total_composite_mean": 0.6446620225906372, "reward_total_composite_std": 0.07332957535982132} {"timestamp_utc": "2026-04-11T23:43:41Z", "mode": "train", "global_step": 1182, "epoch": 0.04747559946981564, "loss": -0.0062, "grad_norm": 2.379884719848633, "learning_rate": 6.4212121212121215e-06, "num_tokens": 2663974.0, "completions/mean_length": 77.5, "completions/min_length": 76.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.5, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9949516654014587, "rewards/meter/std": 0.0005235640564933419, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9949516654014587, "rewards/total_composite/std": 0.0005235640564933419, "reward": 0.9949516654014587, "reward_std": 0.0005235668504610658, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010198801755905151, "sampling/sampling_logp_difference/max": 0.6724367141723633, "sampling/importance_sampling_ratio/min": 0.5104632377624512, "sampling/importance_sampling_ratio/mean": 0.9996510148048401, "sampling/importance_sampling_ratio/max": 1.353779911994934, "entropy": 0.04025013535283506, "clip_ratio/low_mean": 0.008097166079096496, "clip_ratio/low_min": 0.008097166079096496, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.008097166079096496, "reward_total_mean": 0.9949516654014587, "reward_meter_mean": 0.9949516654014587, "reward_meter_std": 0.0005235640564933419, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9949516654014587, "reward_total_composite_std": 0.0005235640564933419} {"timestamp_utc": "2026-04-11T23:43:45Z", "mode": "train", "global_step": 1183, "epoch": 0.04751576495160059, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.418181818181819e-06, "num_tokens": 2665278.0, "completions/mean_length": 28.0, "completions/min_length": 28.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.0, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9889696836471558, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9889696836471558, "rewards/total_composite/std": 0.0, "reward": 0.9889696836471558, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.005411628168076277, "sampling/sampling_logp_difference/max": 0.15753476321697235, "sampling/importance_sampling_ratio/min": 0.8542470932006836, "sampling/importance_sampling_ratio/mean": 0.9997879266738892, "sampling/importance_sampling_ratio/max": 1.0598801374435425, "entropy": 0.046004267409443855, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9889696836471558, "reward_meter_mean": 0.9889696836471558, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9889696836471558, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-11T23:43:50Z", "mode": "train", "global_step": 1184, "epoch": 0.047555930433385546, "loss": 0.038, "grad_norm": 9.193962097167969, "learning_rate": 6.415151515151515e-06, "num_tokens": 2667336.0, "completions/mean_length": 97.25, "completions/min_length": 93.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.25, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.892284095287323, "rewards/meter/std": 0.20539914071559906, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8250000476837158, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.724637508392334, "rewards/total_composite/std": 0.1372780054807663, "reward": 0.724637508392334, "reward_std": 0.1372780054807663, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04928457364439964, "sampling/sampling_logp_difference/max": 1.7316360473632812, "sampling/importance_sampling_ratio/min": 0.17699459195137024, "sampling/importance_sampling_ratio/mean": 0.9922058582305908, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.16915534622967243, "clip_ratio/low_mean": 0.012163461884483695, "clip_ratio/low_min": 0.012163461884483695, "clip_ratio/high_mean": 0.020877854549326003, "clip_ratio/high_max": 0.020877854549326003, "clip_ratio/region_mean": 0.0330413164338097, "reward_total_mean": 0.724637508392334, "reward_meter_mean": 0.892284095287323, "reward_meter_std": 0.20539914071559906, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8250000476837158, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.724637508392334, "reward_total_composite_std": 0.1372780054807663} {"timestamp_utc": "2026-04-11T23:43:55Z", "mode": "train", "global_step": 1185, "epoch": 0.0475960959151705, "loss": -0.0083, "grad_norm": 1.140117883682251, "learning_rate": 6.412121212121213e-06, "num_tokens": 2669230.0, "completions/mean_length": 78.75, "completions/min_length": 78.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.75, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9956543445587158, "rewards/meter/std": 0.0006610880373045802, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9956543445587158, "rewards/total_composite/std": 0.0006610880373045802, "reward": 0.9956543445587158, "reward_std": 0.0006611024145968258, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00999254360795021, "sampling/sampling_logp_difference/max": 0.8007946014404297, "sampling/importance_sampling_ratio/min": 0.44897207617759705, "sampling/importance_sampling_ratio/mean": 1.0024654865264893, "sampling/importance_sampling_ratio/max": 1.5469458103179932, "entropy": 0.040787032805383205, "clip_ratio/low_mean": 0.004807692370377481, "clip_ratio/low_min": 0.004807692370377481, "clip_ratio/high_mean": 0.0031645570416003466, "clip_ratio/high_max": 0.0031645570416003466, "clip_ratio/region_mean": 0.007972249411977828, "reward_total_mean": 0.9956543445587158, "reward_meter_mean": 0.9956543445587158, "reward_meter_std": 0.0006610880373045802, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9956543445587158, "reward_total_composite_std": 0.0006610880373045802} {"timestamp_utc": "2026-04-11T23:44:01Z", "mode": "train", "global_step": 1186, "epoch": 0.047636261396955454, "loss": 0.0014, "grad_norm": 1.235166072845459, "learning_rate": 6.40909090909091e-06, "num_tokens": 2672128.0, "completions/mean_length": 239.25, "completions/min_length": 236.0, "completions/max_length": 242.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 239.25, "completions/min_terminated_length": 236.0, "completions/max_terminated_length": 242.0, "rewards/meter/mean": 0.9895985722541809, "rewards/meter/std": 0.0064008040353655815, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5795454978942871, "rewards/repeat_penalty/std": 0.19399181008338928, "rewards/total_composite/mean": 0.4595944285392761, "rewards/total_composite/std": 0.15526776015758514, "reward": 0.4595944285392761, "reward_std": 0.15526776015758514, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007288929540663958, "sampling/sampling_logp_difference/max": 1.047019124031067, "sampling/importance_sampling_ratio/min": 0.3509824275970459, "sampling/importance_sampling_ratio/mean": 1.0002281665802002, "sampling/importance_sampling_ratio/max": 1.5414388179779053, "entropy": 0.029086438240483403, "clip_ratio/low_mean": 0.0010416667209938169, "clip_ratio/low_min": 0.0010416667209938169, "clip_ratio/high_mean": 0.006786415702663362, "clip_ratio/high_max": 0.006786415702663362, "clip_ratio/region_mean": 0.007828082423657179, "reward_total_mean": 0.4595944285392761, "reward_meter_mean": 0.9895985722541809, "reward_meter_std": 0.0064008040353655815, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5795454978942871, "reward_repeat_penalty_std": 0.19399181008338928, "reward_total_composite_mean": 0.4595944285392761, "reward_total_composite_std": 0.15526776015758514} {"timestamp_utc": "2026-04-11T23:44:10Z", "mode": "train", "global_step": 1187, "epoch": 0.04767642687874041, "loss": 0.0002, "grad_norm": 1.61380934715271, "learning_rate": 6.406060606060607e-06, "num_tokens": 2676850.0, "completions/mean_length": 363.25, "completions/min_length": 348.0, "completions/max_length": 377.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 363.25, "completions/min_terminated_length": 348.0, "completions/max_terminated_length": 377.0, "rewards/meter/mean": 0.9924304485321045, "rewards/meter/std": 0.012585594318807125, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6617647409439087, "rewards/repeat_penalty/std": 0.027230001986026764, "rewards/total_composite/mean": 0.6568292379379272, "rewards/total_composite/std": 0.0302126407623291, "reward": 0.6568292379379272, "reward_std": 0.030212653800845146, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012508058920502663, "sampling/sampling_logp_difference/max": 1.4292956590652466, "sampling/importance_sampling_ratio/min": 0.2921198010444641, "sampling/importance_sampling_ratio/mean": 1.0015922784805298, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06546153873205185, "clip_ratio/low_mean": 0.006156876712338999, "clip_ratio/low_min": 0.006156876712338999, "clip_ratio/high_mean": 0.0030575180426239967, "clip_ratio/high_max": 0.0030575180426239967, "clip_ratio/region_mean": 0.009214394754962996, "reward_total_mean": 0.6568292379379272, "reward_meter_mean": 0.9924304485321045, "reward_meter_std": 0.012585594318807125, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6617647409439087, "reward_repeat_penalty_std": 0.027230001986026764, "reward_total_composite_mean": 0.6568292379379272, "reward_total_composite_std": 0.0302126407623291} {"timestamp_utc": "2026-04-11T23:44:18Z", "mode": "train", "global_step": 1188, "epoch": 0.04771659236052536, "loss": 0.0066, "grad_norm": 1.0509389638900757, "learning_rate": 6.403030303030303e-06, "num_tokens": 2681346.0, "completions/mean_length": 364.0, "completions/min_length": 360.0, "completions/max_length": 368.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 364.0, "completions/min_terminated_length": 360.0, "completions/max_terminated_length": 368.0, "rewards/meter/mean": 0.9938761591911316, "rewards/meter/std": 0.0003768211172427982, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6029411554336548, "rewards/repeat_penalty/std": 0.027230001986026764, "rewards/total_composite/mean": 0.5992562770843506, "rewards/total_composite/std": 0.027259474620223045, "reward": 0.5992562770843506, "reward_std": 0.0272594653069973, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003385041607543826, "sampling/sampling_logp_difference/max": 0.9387906789779663, "sampling/importance_sampling_ratio/min": 0.3911004960536957, "sampling/importance_sampling_ratio/mean": 1.0006787776947021, "sampling/importance_sampling_ratio/max": 1.6698790788650513, "entropy": 0.01585481537040323, "clip_ratio/low_mean": 0.00206068845000118, "clip_ratio/low_min": 0.00206068845000118, "clip_ratio/high_mean": 0.0013888889516238123, "clip_ratio/high_max": 0.0013888889516238123, "clip_ratio/region_mean": 0.0034495774016249925, "reward_total_mean": 0.5992562770843506, "reward_meter_mean": 0.9938761591911316, "reward_meter_std": 0.0003768211172427982, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6029411554336548, "reward_repeat_penalty_std": 0.027230001986026764, "reward_total_composite_mean": 0.5992562770843506, "reward_total_composite_std": 0.027259474620223045} {"timestamp_utc": "2026-04-11T23:44:22Z", "mode": "train", "global_step": 1189, "epoch": 0.047756757842310316, "loss": 0.0367, "grad_norm": 5.781359672546387, "learning_rate": 6.4000000000000006e-06, "num_tokens": 2683179.0, "completions/mean_length": 58.125, "completions/min_length": 56.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.125, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9804081916809082, "rewards/meter/std": 0.011680345050990582, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9804081916809082, "rewards/total_composite/std": 0.011680345050990582, "reward": 0.9804081916809082, "reward_std": 0.011680349707603455, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05228231102228165, "sampling/sampling_logp_difference/max": 2.627631187438965, "sampling/importance_sampling_ratio/min": 0.0722494050860405, "sampling/importance_sampling_ratio/mean": 0.9941091537475586, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15911316499114037, "clip_ratio/low_mean": 0.012431694194674492, "clip_ratio/low_min": 0.012431694194674492, "clip_ratio/high_mean": 0.052866541780531406, "clip_ratio/high_max": 0.052866541780531406, "clip_ratio/region_mean": 0.0652982359752059, "reward_total_mean": 0.9804081916809082, "reward_meter_mean": 0.9804081916809082, "reward_meter_std": 0.011680345050990582, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9804081916809082, "reward_total_composite_std": 0.011680345050990582} {"timestamp_utc": "2026-04-11T23:44:27Z", "mode": "train", "global_step": 1190, "epoch": 0.04779692332409527, "loss": -0.0135, "grad_norm": 6.697084426879883, "learning_rate": 6.396969696969697e-06, "num_tokens": 2685520.0, "completions/mean_length": 113.625, "completions/min_length": 109.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 113.625, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.9840307831764221, "rewards/meter/std": 0.004937502555549145, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8035714626312256, "rewards/repeat_penalty/std": 0.10628911107778549, "rewards/total_composite/mean": 0.7905905246734619, "rewards/total_composite/std": 0.10349281877279282, "reward": 0.7905905246734619, "reward_std": 0.10349280387163162, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023248611018061638, "sampling/sampling_logp_difference/max": 1.2992486953735352, "sampling/importance_sampling_ratio/min": 0.3153161406517029, "sampling/importance_sampling_ratio/mean": 1.004918098449707, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09165592398494482, "clip_ratio/low_mean": 0.010279399110004306, "clip_ratio/low_min": 0.010279399110004306, "clip_ratio/high_mean": 0.008562042959965765, "clip_ratio/high_max": 0.008562042959965765, "clip_ratio/region_mean": 0.01884144206997007, "reward_total_mean": 0.7905905246734619, "reward_meter_mean": 0.9840307831764221, "reward_meter_std": 0.004937502555549145, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8035714626312256, "reward_repeat_penalty_std": 0.10628911107778549, "reward_total_composite_mean": 0.7905905246734619, "reward_total_composite_std": 0.10349281877279282} {"timestamp_utc": "2026-04-11T23:44:32Z", "mode": "train", "global_step": 1191, "epoch": 0.047837088805880223, "loss": -0.0018, "grad_norm": 3.869032859802246, "learning_rate": 6.393939393939394e-06, "num_tokens": 2687846.0, "completions/mean_length": 117.75, "completions/min_length": 113.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 117.75, "completions/min_terminated_length": 113.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.9962882399559021, "rewards/meter/std": 0.002757209585979581, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.8966987133026123, "rewards/total_composite/std": 0.10691322386264801, "reward": 0.8966987133026123, "reward_std": 0.10691321641206741, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03010808303952217, "sampling/sampling_logp_difference/max": 1.3976101875305176, "sampling/importance_sampling_ratio/min": 0.24718700349330902, "sampling/importance_sampling_ratio/mean": 1.0035390853881836, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.16953753679990768, "clip_ratio/low_mean": 0.013930454850196838, "clip_ratio/low_min": 0.013930454850196838, "clip_ratio/high_mean": 0.01574227074161172, "clip_ratio/high_max": 0.01574227074161172, "clip_ratio/region_mean": 0.029672725591808558, "reward_total_mean": 0.8966987133026123, "reward_meter_mean": 0.9962882399559021, "reward_meter_std": 0.002757209585979581, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_total_composite_mean": 0.8966987133026123, "reward_total_composite_std": 0.10691322386264801} {"timestamp_utc": "2026-04-11T23:44:37Z", "mode": "train", "global_step": 1192, "epoch": 0.04787725428766518, "loss": -0.0028, "grad_norm": 5.973384380340576, "learning_rate": 6.390909090909091e-06, "num_tokens": 2689734.0, "completions/mean_length": 79.0, "completions/min_length": 79.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.0, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.994601309299469, "rewards/meter/std": 0.003018211107701063, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.994601309299469, "rewards/total_composite/std": 0.003018211107701063, "reward": 0.994601309299469, "reward_std": 0.0030182143673300743, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01802201196551323, "sampling/sampling_logp_difference/max": 1.7237977981567383, "sampling/importance_sampling_ratio/min": 0.3428289294242859, "sampling/importance_sampling_ratio/mean": 1.0011334419250488, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.033365981886163354, "clip_ratio/low_mean": 0.0015822785208001733, "clip_ratio/low_min": 0.0015822785208001733, "clip_ratio/high_mean": 0.007911392254754901, "clip_ratio/high_max": 0.007911392254754901, "clip_ratio/region_mean": 0.009493670775555074, "reward_total_mean": 0.994601309299469, "reward_meter_mean": 0.994601309299469, "reward_meter_std": 0.003018211107701063, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.994601309299469, "reward_total_composite_std": 0.003018211107701063} {"timestamp_utc": "2026-04-11T23:44:41Z", "mode": "train", "global_step": 1193, "epoch": 0.04791741976945013, "loss": -0.0592, "grad_norm": 7.629351615905762, "learning_rate": 6.387878787878789e-06, "num_tokens": 2691559.0, "completions/mean_length": 61.125, "completions/min_length": 49.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.125, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.8711988925933838, "rewards/meter/std": 0.3447757959365845, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8711988925933838, "rewards/total_composite/std": 0.3447757959365845, "reward": 0.8711988925933838, "reward_std": 0.3447757661342621, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03702111169695854, "sampling/sampling_logp_difference/max": 0.9225618839263916, "sampling/importance_sampling_ratio/min": 0.3974994122982025, "sampling/importance_sampling_ratio/mean": 1.0055021047592163, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.16935504972934723, "clip_ratio/low_mean": 0.015306122601032257, "clip_ratio/low_min": 0.015306122601032257, "clip_ratio/high_mean": 0.027820809744298458, "clip_ratio/high_max": 0.027820809744298458, "clip_ratio/region_mean": 0.043126932345330715, "reward_total_mean": 0.8711988925933838, "reward_meter_mean": 0.8711988925933838, "reward_meter_std": 0.3447757959365845, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8711988925933838, "reward_total_composite_std": 0.3447757959365845} {"timestamp_utc": "2026-04-11T23:44:46Z", "mode": "train", "global_step": 1194, "epoch": 0.047957585251235085, "loss": 0.0146, "grad_norm": 4.607259750366211, "learning_rate": 6.384848484848485e-06, "num_tokens": 2693468.0, "completions/mean_length": 83.625, "completions/min_length": 81.0, "completions/max_length": 89.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.625, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 89.0, "rewards/meter/mean": 0.9963547587394714, "rewards/meter/std": 0.0017178525449708104, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9963547587394714, "rewards/total_composite/std": 0.0017178525449708104, "reward": 0.9963547587394714, "reward_std": 0.0017178438138216734, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0324261412024498, "sampling/sampling_logp_difference/max": 1.778012990951538, "sampling/importance_sampling_ratio/min": 0.16897356510162354, "sampling/importance_sampling_ratio/mean": 1.0033197402954102, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12350557837635279, "clip_ratio/low_mean": 0.0029069767333567142, "clip_ratio/low_min": 0.0029069767333567142, "clip_ratio/high_mean": 0.02205510064959526, "clip_ratio/high_max": 0.02205510064959526, "clip_ratio/region_mean": 0.024962077382951975, "reward_total_mean": 0.9963547587394714, "reward_meter_mean": 0.9963547587394714, "reward_meter_std": 0.0017178525449708104, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9963547587394714, "reward_total_composite_std": 0.0017178525449708104} {"timestamp_utc": "2026-04-11T23:44:50Z", "mode": "train", "global_step": 1195, "epoch": 0.04799775073302004, "loss": -0.0016, "grad_norm": 6.79469108581543, "learning_rate": 6.381818181818182e-06, "num_tokens": 2695188.0, "completions/mean_length": 56.0, "completions/min_length": 55.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.0, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.08778245747089386, "rewards/meter/std": 0.11504703015089035, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8333333730697632, "rewards/repeat_penalty/std": 0.17817415297031403, "rewards/total_composite/mean": 0.07255840301513672, "rewards/total_composite/std": 0.08352638781070709, "reward": 0.07255840301513672, "reward_std": 0.0835263803601265, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022929562255740166, "sampling/sampling_logp_difference/max": 0.9827042818069458, "sampling/importance_sampling_ratio/min": 0.37429752945899963, "sampling/importance_sampling_ratio/mean": 1.0030452013015747, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11227265931665897, "clip_ratio/low_mean": 0.006657268386334181, "clip_ratio/low_min": 0.006657268386334181, "clip_ratio/high_mean": 0.011043233331292868, "clip_ratio/high_max": 0.011043233331292868, "clip_ratio/region_mean": 0.01770050171762705, "reward_total_mean": 0.07255840301513672, "reward_meter_mean": 0.08778245747089386, "reward_meter_std": 0.11504703015089035, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8333333730697632, "reward_repeat_penalty_std": 0.17817415297031403, "reward_total_composite_mean": 0.07255840301513672, "reward_total_composite_std": 0.08352638781070709} {"timestamp_utc": "2026-04-11T23:44:55Z", "mode": "train", "global_step": 1196, "epoch": 0.04803791621480499, "loss": 0.0143, "grad_norm": 5.267541408538818, "learning_rate": 6.37878787878788e-06, "num_tokens": 2697251.0, "completions/mean_length": 92.875, "completions/min_length": 91.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.875, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.9628203511238098, "rewards/meter/std": 0.07420913130044937, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.8394116759300232, "rewards/total_composite/std": 0.0950227826833725, "reward": 0.8394116759300232, "reward_std": 0.0950227826833725, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.046500932425260544, "sampling/sampling_logp_difference/max": 1.5942132472991943, "sampling/importance_sampling_ratio/min": 0.20306822657585144, "sampling/importance_sampling_ratio/mean": 0.9947278499603271, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.14395452290773392, "clip_ratio/low_mean": 0.023972635506652296, "clip_ratio/low_min": 0.023972635506652296, "clip_ratio/high_mean": 0.012287983670830727, "clip_ratio/high_max": 0.012287983670830727, "clip_ratio/region_mean": 0.03626061917748302, "reward_total_mean": 0.8394116759300232, "reward_meter_mean": 0.9628203511238098, "reward_meter_std": 0.07420913130044937, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.8394116759300232, "reward_total_composite_std": 0.0950227826833725} {"timestamp_utc": "2026-04-11T23:45:05Z", "mode": "train", "global_step": 1197, "epoch": 0.048078081696589954, "loss": 0.0431, "grad_norm": 0.9943876266479492, "learning_rate": 6.375757575757576e-06, "num_tokens": 2702278.0, "completions/mean_length": 419.375, "completions/min_length": 402.0, "completions/max_length": 474.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 419.375, "completions/min_terminated_length": 402.0, "completions/max_terminated_length": 474.0, "rewards/meter/mean": 0.8696904182434082, "rewards/meter/std": 0.3263057768344879, "rewards/count_adherence/mean": 0.6333333253860474, "rewards/count_adherence/std": 0.0942809134721756, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6529605388641357, "rewards/repeat_penalty/std": 0.05881762132048607, "rewards/total_composite/mean": 0.3697792887687683, "rewards/total_composite/std": 0.14553934335708618, "reward": 0.3697792887687683, "reward_std": 0.14553934335708618, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009604532271623611, "sampling/sampling_logp_difference/max": 1.1023553609848022, "sampling/importance_sampling_ratio/min": 0.3320879638195038, "sampling/importance_sampling_ratio/mean": 1.0019516944885254, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.04937975201755762, "clip_ratio/low_mean": 0.0029802137287333608, "clip_ratio/low_min": 0.0029802137287333608, "clip_ratio/high_mean": 0.004789536877069622, "clip_ratio/high_max": 0.004789536877069622, "clip_ratio/region_mean": 0.007769750605802983, "reward_total_mean": 0.3697792887687683, "reward_meter_mean": 0.8696904182434082, "reward_meter_std": 0.3263057768344879, "reward_count_adherence_mean": 0.6333333253860474, "reward_count_adherence_std": 0.0942809134721756, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6529605388641357, "reward_repeat_penalty_std": 0.05881762132048607, "reward_total_composite_mean": 0.3697792887687683, "reward_total_composite_std": 0.14553934335708618} {"timestamp_utc": "2026-04-11T23:45:09Z", "mode": "train", "global_step": 1198, "epoch": 0.04811824717837491, "loss": 0.0032, "grad_norm": 6.456137657165527, "learning_rate": 6.372727272727274e-06, "num_tokens": 2703954.0, "completions/mean_length": 61.5, "completions/min_length": 59.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.5, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.3377665877342224, "rewards/meter/std": 0.33047032356262207, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.3377665877342224, "rewards/total_composite/std": 0.33047032356262207, "reward": 0.3377665877342224, "reward_std": 0.33047032356262207, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0402526780962944, "sampling/sampling_logp_difference/max": 1.6591825485229492, "sampling/importance_sampling_ratio/min": 0.19029447436332703, "sampling/importance_sampling_ratio/mean": 1.0035641193389893, "sampling/importance_sampling_ratio/max": 1.5698946714401245, "entropy": 0.31682545877993107, "clip_ratio/low_mean": 0.014116051141172647, "clip_ratio/low_min": 0.014116051141172647, "clip_ratio/high_mean": 0.006082487525418401, "clip_ratio/high_max": 0.006082487525418401, "clip_ratio/region_mean": 0.020198538666591048, "reward_total_mean": 0.3377665877342224, "reward_meter_mean": 0.3377665877342224, "reward_meter_std": 0.33047032356262207, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.3377665877342224, "reward_total_composite_std": 0.33047032356262207} {"timestamp_utc": "2026-04-11T23:45:14Z", "mode": "train", "global_step": 1199, "epoch": 0.04815841266015986, "loss": 0.0199, "grad_norm": 4.539976596832275, "learning_rate": 6.3696969696969706e-06, "num_tokens": 2705923.0, "completions/mean_length": 81.125, "completions/min_length": 79.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.125, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.5489683747291565, "rewards/meter/std": 0.4779263734817505, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5489683747291565, "rewards/total_composite/std": 0.4779263734817505, "reward": 0.5489683747291565, "reward_std": 0.4779263436794281, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010573667474091053, "sampling/sampling_logp_difference/max": 0.9852040410041809, "sampling/importance_sampling_ratio/min": 0.37336301803588867, "sampling/importance_sampling_ratio/mean": 1.0014384984970093, "sampling/importance_sampling_ratio/max": 1.4673889875411987, "entropy": 0.04470323724672198, "clip_ratio/low_mean": 0.004537329194135964, "clip_ratio/low_min": 0.004537329194135964, "clip_ratio/high_mean": 0.0031447785440832376, "clip_ratio/high_max": 0.0031447785440832376, "clip_ratio/region_mean": 0.0076821077382192016, "reward_total_mean": 0.5489683747291565, "reward_meter_mean": 0.5489683747291565, "reward_meter_std": 0.4779263734817505, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5489683747291565, "reward_total_composite_std": 0.4779263734817505} {"timestamp_utc": "2026-04-11T23:45:18Z", "mode": "train", "global_step": 1200, "epoch": 0.048198578141944816, "loss": 0.018, "grad_norm": 13.826656341552734, "learning_rate": 6.366666666666668e-06, "num_tokens": 2707810.0, "completions/mean_length": 59.875, "completions/min_length": 56.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.875, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.8469290137290955, "rewards/meter/std": 0.20266500115394592, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8469290137290955, "rewards/total_composite/std": 0.20266500115394592, "reward": 0.8469290137290955, "reward_std": 0.20266498625278473, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.036583464592695236, "sampling/sampling_logp_difference/max": 1.854140043258667, "sampling/importance_sampling_ratio/min": 0.15658754110336304, "sampling/importance_sampling_ratio/mean": 1.0088112354278564, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09853810910135508, "clip_ratio/low_mean": 0.0147115015424788, "clip_ratio/low_min": 0.0147115015424788, "clip_ratio/high_mean": 0.0020491802133619785, "clip_ratio/high_max": 0.0020491802133619785, "clip_ratio/region_mean": 0.01676068175584078, "reward_total_mean": 0.8469290137290955, "reward_meter_mean": 0.8469290137290955, "reward_meter_std": 0.20266500115394592, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8469290137290955, "reward_total_composite_std": 0.20266500115394592} {"timestamp_utc": "2026-04-11T23:46:27Z", "mode": "eval", "global_step": 1200, "epoch": 0.048198578141944816, "eval_loss": NaN, "eval_runtime": 69.5055, "eval_samples_per_second": 1.496, "eval_steps_per_second": 0.187, "eval_num_tokens": 2707810.0, "eval_completions/mean_length": 210.4903846153846, "eval_completions/min_length": 63.69230769230769, "eval_completions/max_length": 365.9230769230769, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 210.4903846153846, "eval_completions/min_terminated_length": 63.69230769230769, "eval_completions/max_terminated_length": 365.9230769230769, "eval_rewards/meter/mean": 0.675024688243866, "eval_rewards/meter/std": 0.4085804086465102, "eval_rewards/count_adherence/mean": 0.8891764970926138, "eval_rewards/count_adherence/std": 0.11591204485067955, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/repeat_penalty/mean": 0.7145149661944463, "eval_rewards/repeat_penalty/std": 0.17899919931705183, "eval_rewards/total_composite/mean": 0.4327001617505, "eval_rewards/total_composite/std": 0.3292730886202592, "eval_reward": 0.4327001617505, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.009530584721897658, "eval_sampling/sampling_logp_difference/max": 0.8752486155583308, "eval_sampling/importance_sampling_ratio/min": 0.4270020562868852, "eval_sampling/importance_sampling_ratio/mean": 1.0017014145851135, "eval_sampling/importance_sampling_ratio/max": 1.3788938980836134, "eval_entropy": 0.07712389929936482, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.4327001617505, "eval_reward_meter_mean": 0.675024688243866, "eval_reward_meter_std": 0.4085804086465102, "eval_reward_count_adherence_mean": 0.8891764970926138, "eval_reward_count_adherence_std": 0.11591204485067955, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_repeat_penalty_mean": 0.7145149661944463, "eval_reward_repeat_penalty_std": 0.17899919931705183, "eval_reward_total_composite_mean": 0.4327001617505, "eval_reward_total_composite_std": 0.3292730886202592} {"timestamp_utc": "2026-04-11T23:46:35Z", "mode": "train", "global_step": 1201, "epoch": 0.04823874362372977, "loss": 0.0014, "grad_norm": 1.0187959671020508, "learning_rate": 6.363636363636364e-06, "num_tokens": 2709651.0, "completions/mean_length": 79.125, "completions/min_length": 79.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.125, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9956833124160767, "rewards/meter/std": 4.2210078390780836e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9956833124160767, "rewards/total_composite/std": 4.2210078390780836e-05, "reward": 0.9956833124160767, "reward_std": 4.2219100578222424e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0033457186073064804, "sampling/sampling_logp_difference/max": 0.5514485836029053, "sampling/importance_sampling_ratio/min": 0.5761146545410156, "sampling/importance_sampling_ratio/mean": 1.0009052753448486, "sampling/importance_sampling_ratio/max": 1.1379395723342896, "entropy": 0.022257187170907855, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0015822785208001733, "clip_ratio/high_max": 0.0015822785208001733, "clip_ratio/region_mean": 0.0015822785208001733, "reward_total_mean": 0.9956833124160767, "reward_meter_mean": 0.9956833124160767, "reward_meter_std": 4.2210078390780836e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9956833124160767, "reward_total_composite_std": 4.2210078390780836e-05} {"timestamp_utc": "2026-04-11T23:46:40Z", "mode": "train", "global_step": 1202, "epoch": 0.048278909105514724, "loss": -0.0406, "grad_norm": 8.3601655960083, "learning_rate": 6.3606060606060615e-06, "num_tokens": 2711138.0, "completions/mean_length": 29.875, "completions/min_length": 27.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.875, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.644133985042572, "rewards/meter/std": 0.4584461748600006, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.644133985042572, "rewards/total_composite/std": 0.4584461748600006, "reward": 0.644133985042572, "reward_std": 0.4584461748600006, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.034943368285894394, "sampling/sampling_logp_difference/max": 1.0919749736785889, "sampling/importance_sampling_ratio/min": 0.3355531394481659, "sampling/importance_sampling_ratio/mean": 1.0082812309265137, "sampling/importance_sampling_ratio/max": 1.6411713361740112, "entropy": 0.21921741589903831, "clip_ratio/low_mean": 0.03160349419340491, "clip_ratio/low_min": 0.03160349419340491, "clip_ratio/high_mean": 0.012096773833036423, "clip_ratio/high_max": 0.012096773833036423, "clip_ratio/region_mean": 0.043700268026441336, "reward_total_mean": 0.644133985042572, "reward_meter_mean": 0.644133985042572, "reward_meter_std": 0.4584461748600006, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.644133985042572, "reward_total_composite_std": 0.4584461748600006} {"timestamp_utc": "2026-04-11T23:46:45Z", "mode": "train", "global_step": 1203, "epoch": 0.04831907458729968, "loss": 0.0002, "grad_norm": 4.518101215362549, "learning_rate": 6.357575757575758e-06, "num_tokens": 2712910.0, "completions/mean_length": 60.5, "completions/min_length": 59.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.5, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.203798308968544, "rewards/meter/std": 0.2958202362060547, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.203798308968544, "rewards/total_composite/std": 0.2958202362060547, "reward": 0.203798308968544, "reward_std": 0.2958202362060547, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05345018953084946, "sampling/sampling_logp_difference/max": 1.703664779663086, "sampling/importance_sampling_ratio/min": 0.18201525509357452, "sampling/importance_sampling_ratio/mean": 1.004960060119629, "sampling/importance_sampling_ratio/max": 1.886362910270691, "entropy": 0.37029214575886726, "clip_ratio/low_mean": 0.04342324007302523, "clip_ratio/low_min": 0.04342324007302523, "clip_ratio/high_mean": 0.008405112195760012, "clip_ratio/high_max": 0.008405112195760012, "clip_ratio/region_mean": 0.05182835226878524, "reward_total_mean": 0.203798308968544, "reward_meter_mean": 0.203798308968544, "reward_meter_std": 0.2958202362060547, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.203798308968544, "reward_total_composite_std": 0.2958202362060547} {"timestamp_utc": "2026-04-11T23:46:50Z", "mode": "train", "global_step": 1204, "epoch": 0.04835924006908463, "loss": 0.004, "grad_norm": 3.396805763244629, "learning_rate": 6.354545454545455e-06, "num_tokens": 2715103.0, "completions/mean_length": 94.125, "completions/min_length": 91.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.125, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.9817042350769043, "rewards/meter/std": 0.002063595922663808, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8250000476837158, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8099536895751953, "rewards/total_composite/std": 0.070185586810112, "reward": 0.8099536895751953, "reward_std": 0.070185586810112, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022159146144986153, "sampling/sampling_logp_difference/max": 1.2238504886627197, "sampling/importance_sampling_ratio/min": 0.29409554600715637, "sampling/importance_sampling_ratio/mean": 1.0024614334106445, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06619423814117908, "clip_ratio/low_mean": 0.01712075702380389, "clip_ratio/low_min": 0.01712075702380389, "clip_ratio/high_mean": 0.003989361692219973, "clip_ratio/high_max": 0.003989361692219973, "clip_ratio/region_mean": 0.021110118716023862, "reward_total_mean": 0.8099536895751953, "reward_meter_mean": 0.9817042350769043, "reward_meter_std": 0.002063595922663808, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8250000476837158, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.8099536895751953, "reward_total_composite_std": 0.070185586810112} {"timestamp_utc": "2026-04-11T23:46:55Z", "mode": "train", "global_step": 1205, "epoch": 0.048399405550869586, "loss": 0.0175, "grad_norm": 4.248809814453125, "learning_rate": 6.3515151515151516e-06, "num_tokens": 2717019.0, "completions/mean_length": 78.5, "completions/min_length": 75.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.5, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.9937853813171387, "rewards/meter/std": 0.0054353163577616215, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9937853813171387, "rewards/total_composite/std": 0.0054353163577616215, "reward": 0.9937853813171387, "reward_std": 0.0054353224113583565, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03144395351409912, "sampling/sampling_logp_difference/max": 1.310187816619873, "sampling/importance_sampling_ratio/min": 0.2697693705558777, "sampling/importance_sampling_ratio/mean": 1.0043216943740845, "sampling/importance_sampling_ratio/max": 1.6284278631210327, "entropy": 0.18105985596776009, "clip_ratio/low_mean": 0.004518072120845318, "clip_ratio/low_min": 0.004518072120845318, "clip_ratio/high_mean": 0.014675587648525834, "clip_ratio/high_max": 0.014675587648525834, "clip_ratio/region_mean": 0.019193659769371152, "reward_total_mean": 0.9937853813171387, "reward_meter_mean": 0.9937853813171387, "reward_meter_std": 0.0054353163577616215, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9937853813171387, "reward_total_composite_std": 0.0054353163577616215} {"timestamp_utc": "2026-04-11T23:47:00Z", "mode": "train", "global_step": 1206, "epoch": 0.04843957103265454, "loss": -0.0049, "grad_norm": 2.2625913619995117, "learning_rate": 6.34848484848485e-06, "num_tokens": 2718833.0, "completions/mean_length": 77.75, "completions/min_length": 76.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.75, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9951738119125366, "rewards/meter/std": 0.00032568458118475974, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9951738119125366, "rewards/total_composite/std": 0.00032568458118475974, "reward": 0.9951738119125366, "reward_std": 0.00032568458118475974, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004397572483867407, "sampling/sampling_logp_difference/max": 0.4512134790420532, "sampling/importance_sampling_ratio/min": 0.636854887008667, "sampling/importance_sampling_ratio/mean": 1.0006308555603027, "sampling/importance_sampling_ratio/max": 1.3814133405685425, "entropy": 0.022122491151094437, "clip_ratio/low_mean": 0.001623376621864736, "clip_ratio/low_min": 0.001623376621864736, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.001623376621864736, "reward_total_mean": 0.9951738119125366, "reward_meter_mean": 0.9951738119125366, "reward_meter_std": 0.00032568458118475974, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9951738119125366, "reward_total_composite_std": 0.00032568458118475974} {"timestamp_utc": "2026-04-11T23:47:06Z", "mode": "train", "global_step": 1207, "epoch": 0.04847973651443949, "loss": -0.005, "grad_norm": 7.367405891418457, "learning_rate": 6.345454545454546e-06, "num_tokens": 2721333.0, "completions/mean_length": 124.5, "completions/min_length": 114.0, "completions/max_length": 130.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 124.5, "completions/min_terminated_length": 114.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.8035296201705933, "rewards/meter/std": 0.35723742842674255, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.5791441202163696, "rewards/total_composite/std": 0.24694091081619263, "reward": 0.5791441202163696, "reward_std": 0.24694089591503143, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022883908823132515, "sampling/sampling_logp_difference/max": 1.5613296031951904, "sampling/importance_sampling_ratio/min": 0.20985685288906097, "sampling/importance_sampling_ratio/mean": 1.0008379220962524, "sampling/importance_sampling_ratio/max": 1.7591142654418945, "entropy": 0.1076383925974369, "clip_ratio/low_mean": 0.005122669972479343, "clip_ratio/low_min": 0.005122669972479343, "clip_ratio/high_mean": 0.013700352283194661, "clip_ratio/high_max": 0.013700352283194661, "clip_ratio/region_mean": 0.018823022255674005, "reward_total_mean": 0.5791441202163696, "reward_meter_mean": 0.8035296201705933, "reward_meter_std": 0.35723742842674255, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.5791441202163696, "reward_total_composite_std": 0.24694091081619263} {"timestamp_utc": "2026-04-11T23:47:10Z", "mode": "train", "global_step": 1208, "epoch": 0.04851990199622445, "loss": -0.0147, "grad_norm": 5.575094699859619, "learning_rate": 6.342424242424243e-06, "num_tokens": 2723045.0, "completions/mean_length": 63.0, "completions/min_length": 55.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.0, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.3075833320617676, "rewards/meter/std": 0.3925836980342865, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.3075833320617676, "rewards/total_composite/std": 0.3925836980342865, "reward": 0.3075833320617676, "reward_std": 0.3925836980342865, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07028155028820038, "sampling/sampling_logp_difference/max": 1.5979199409484863, "sampling/importance_sampling_ratio/min": 0.20231691002845764, "sampling/importance_sampling_ratio/mean": 0.9958197474479675, "sampling/importance_sampling_ratio/max": 1.9152469635009766, "entropy": 0.3939699362963438, "clip_ratio/low_mean": 0.03671912616118789, "clip_ratio/low_min": 0.03671912616118789, "clip_ratio/high_mean": 0.016762672923505306, "clip_ratio/high_max": 0.016762672923505306, "clip_ratio/region_mean": 0.05348179908469319, "reward_total_mean": 0.3075833320617676, "reward_meter_mean": 0.3075833320617676, "reward_meter_std": 0.3925836980342865, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.3075833320617676, "reward_total_composite_std": 0.3925836980342865} {"timestamp_utc": "2026-04-11T23:47:18Z", "mode": "train", "global_step": 1209, "epoch": 0.0485600674780094, "loss": -0.0039, "grad_norm": 2.477656364440918, "learning_rate": 6.33939393939394e-06, "num_tokens": 2726311.0, "completions/mean_length": 230.25, "completions/min_length": 220.0, "completions/max_length": 246.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 230.25, "completions/min_terminated_length": 220.0, "completions/max_terminated_length": 246.0, "rewards/meter/mean": 0.9955352544784546, "rewards/meter/std": 0.0017380027566105127, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6704545617103577, "rewards/repeat_penalty/std": 0.13690368831157684, "rewards/total_composite/mean": 0.5340035557746887, "rewards/total_composite/std": 0.10914173722267151, "reward": 0.5340035557746887, "reward_std": 0.10914173722267151, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020786819979548454, "sampling/sampling_logp_difference/max": 1.8532099723815918, "sampling/importance_sampling_ratio/min": 0.15673324465751648, "sampling/importance_sampling_ratio/mean": 1.0021883249282837, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10937195271253586, "clip_ratio/low_mean": 0.004963513347320259, "clip_ratio/low_min": 0.004963513347320259, "clip_ratio/high_mean": 0.010197082534432411, "clip_ratio/high_max": 0.010197082534432411, "clip_ratio/region_mean": 0.01516059588175267, "reward_total_mean": 0.5340035557746887, "reward_meter_mean": 0.9955352544784546, "reward_meter_std": 0.0017380027566105127, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6704545617103577, "reward_repeat_penalty_std": 0.13690368831157684, "reward_total_composite_mean": 0.5340035557746887, "reward_total_composite_std": 0.10914173722267151} {"timestamp_utc": "2026-04-11T23:47:24Z", "mode": "train", "global_step": 1210, "epoch": 0.048600232959794355, "loss": 0.0075, "grad_norm": 0.6899715662002563, "learning_rate": 6.336363636363637e-06, "num_tokens": 2729552.0, "completions/mean_length": 232.125, "completions/min_length": 228.0, "completions/max_length": 235.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 232.125, "completions/min_terminated_length": 228.0, "completions/max_terminated_length": 235.0, "rewards/meter/mean": 0.9961996078491211, "rewards/meter/std": 0.0006288865115493536, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4545454680919647, "rewards/repeat_penalty/std": 0.11902794241905212, "rewards/total_composite/mean": 0.36228621006011963, "rewards/total_composite/std": 0.09505681693553925, "reward": 0.36228621006011963, "reward_std": 0.09505681693553925, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005651684943586588, "sampling/sampling_logp_difference/max": 0.983978271484375, "sampling/importance_sampling_ratio/min": 0.3738209903240204, "sampling/importance_sampling_ratio/mean": 1.0020835399627686, "sampling/importance_sampling_ratio/max": 1.5110461711883545, "entropy": 0.03385857096873224, "clip_ratio/low_mean": 0.0016003104392439127, "clip_ratio/low_min": 0.0016003104392439127, "clip_ratio/high_mean": 0.001087038719560951, "clip_ratio/high_max": 0.001087038719560951, "clip_ratio/region_mean": 0.0026873491588048637, "reward_total_mean": 0.36228621006011963, "reward_meter_mean": 0.9961996078491211, "reward_meter_std": 0.0006288865115493536, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4545454680919647, "reward_repeat_penalty_std": 0.11902794241905212, "reward_total_composite_mean": 0.36228621006011963, "reward_total_composite_std": 0.09505681693553925} {"timestamp_utc": "2026-04-11T23:47:32Z", "mode": "train", "global_step": 1211, "epoch": 0.04864039844157931, "loss": 0.0165, "grad_norm": 2.939035177230835, "learning_rate": 6.333333333333333e-06, "num_tokens": 2733597.0, "completions/mean_length": 305.625, "completions/min_length": 289.0, "completions/max_length": 320.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 305.625, "completions/min_terminated_length": 289.0, "completions/max_terminated_length": 320.0, "rewards/meter/mean": 0.9956304430961609, "rewards/meter/std": 0.0010864537907764316, "rewards/count_adherence/mean": 0.9444444179534912, "rewards/count_adherence/std": 0.059391383081674576, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7098382711410522, "rewards/repeat_penalty/std": 0.09350526332855225, "rewards/total_composite/mean": 0.6673710942268372, "rewards/total_composite/std": 0.0976538211107254, "reward": 0.6673710942268372, "reward_std": 0.0976538211107254, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01374336052685976, "sampling/sampling_logp_difference/max": 2.966305732727051, "sampling/importance_sampling_ratio/min": 0.06939523667097092, "sampling/importance_sampling_ratio/mean": 1.001654028892517, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.04799028439447284, "clip_ratio/low_mean": 0.00556039166986011, "clip_ratio/low_min": 0.00556039166986011, "clip_ratio/high_mean": 0.005265707382932305, "clip_ratio/high_max": 0.005265707382932305, "clip_ratio/region_mean": 0.010826099052792415, "reward_total_mean": 0.6673710942268372, "reward_meter_mean": 0.9956304430961609, "reward_meter_std": 0.0010864537907764316, "reward_count_adherence_mean": 0.9444444179534912, "reward_count_adherence_std": 0.059391383081674576, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7098382711410522, "reward_repeat_penalty_std": 0.09350526332855225, "reward_total_composite_mean": 0.6673710942268372, "reward_total_composite_std": 0.0976538211107254} {"timestamp_utc": "2026-04-11T23:47:38Z", "mode": "train", "global_step": 1212, "epoch": 0.04868056392336426, "loss": 0.0002, "grad_norm": 1.3120684623718262, "learning_rate": 6.330303030303031e-06, "num_tokens": 2735462.0, "completions/mean_length": 62.125, "completions/min_length": 61.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.125, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.993018627166748, "rewards/meter/std": 0.0011156242107972503, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.993018627166748, "rewards/total_composite/std": 0.0011156242107972503, "reward": 0.993018627166748, "reward_std": 0.0011156280525028706, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016475966200232506, "sampling/sampling_logp_difference/max": 1.2096130847930908, "sampling/importance_sampling_ratio/min": 0.4961382746696472, "sampling/importance_sampling_ratio/mean": 1.0012403726577759, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.05741153517737985, "clip_ratio/low_mean": 0.010017690248787403, "clip_ratio/low_min": 0.010017690248787403, "clip_ratio/high_mean": 0.004032257944345474, "clip_ratio/high_max": 0.004032257944345474, "clip_ratio/region_mean": 0.014049948193132877, "reward_total_mean": 0.993018627166748, "reward_meter_mean": 0.993018627166748, "reward_meter_std": 0.0011156242107972503, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.993018627166748, "reward_total_composite_std": 0.0011156242107972503} {"timestamp_utc": "2026-04-11T23:47:43Z", "mode": "train", "global_step": 1213, "epoch": 0.04872072940514922, "loss": -0.0134, "grad_norm": 5.671288967132568, "learning_rate": 6.327272727272727e-06, "num_tokens": 2737205.0, "completions/mean_length": 61.875, "completions/min_length": 59.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.875, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9939371347427368, "rewards/meter/std": 0.0013982808450236917, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9939371347427368, "rewards/total_composite/std": 0.0013982808450236917, "reward": 0.9939371347427368, "reward_std": 0.0013982723467051983, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02799602597951889, "sampling/sampling_logp_difference/max": 1.1926932334899902, "sampling/importance_sampling_ratio/min": 0.3034030497074127, "sampling/importance_sampling_ratio/mean": 0.9959902763366699, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08121698070317507, "clip_ratio/low_mean": 0.012202381622046232, "clip_ratio/low_min": 0.012202381622046232, "clip_ratio/high_mean": 0.007692307815887034, "clip_ratio/high_max": 0.007692307815887034, "clip_ratio/region_mean": 0.019894689437933266, "reward_total_mean": 0.9939371347427368, "reward_meter_mean": 0.9939371347427368, "reward_meter_std": 0.0013982808450236917, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9939371347427368, "reward_total_composite_std": 0.0013982808450236917} {"timestamp_utc": "2026-04-11T23:47:52Z", "mode": "train", "global_step": 1214, "epoch": 0.04876089488693417, "loss": -0.0101, "grad_norm": 1.0214275121688843, "learning_rate": 6.324242424242425e-06, "num_tokens": 2741692.0, "completions/mean_length": 323.875, "completions/min_length": 316.0, "completions/max_length": 362.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 323.875, "completions/min_terminated_length": 316.0, "completions/max_terminated_length": 362.0, "rewards/meter/mean": 0.9967888593673706, "rewards/meter/std": 0.0008384960819967091, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.035355325788259506, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6199448704719543, "rewards/repeat_penalty/std": 0.08540944010019302, "rewards/total_composite/mean": 0.5024136304855347, "rewards/total_composite/std": 0.07446450740098953, "reward": 0.5024136304855347, "reward_std": 0.07446449995040894, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007462342269718647, "sampling/sampling_logp_difference/max": 2.224062919616699, "sampling/importance_sampling_ratio/min": 0.10816873610019684, "sampling/importance_sampling_ratio/mean": 1.0000638961791992, "sampling/importance_sampling_ratio/max": 1.6564884185791016, "entropy": 0.0330718404147774, "clip_ratio/low_mean": 0.0023547938617412, "clip_ratio/low_min": 0.0023547938617412, "clip_ratio/high_mean": 0.002272886282298714, "clip_ratio/high_max": 0.002272886282298714, "clip_ratio/region_mean": 0.004627680144039914, "reward_total_mean": 0.5024136304855347, "reward_meter_mean": 0.9967888593673706, "reward_meter_std": 0.0008384960819967091, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.035355325788259506, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6199448704719543, "reward_repeat_penalty_std": 0.08540944010019302, "reward_total_composite_mean": 0.5024136304855347, "reward_total_composite_std": 0.07446450740098953} {"timestamp_utc": "2026-04-11T23:48:00Z", "mode": "train", "global_step": 1215, "epoch": 0.048801060368719125, "loss": 0.0064, "grad_norm": 2.1483280658721924, "learning_rate": 6.3212121212121216e-06, "num_tokens": 2745177.0, "completions/mean_length": 241.625, "completions/min_length": 235.0, "completions/max_length": 246.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 241.625, "completions/min_terminated_length": 235.0, "completions/max_terminated_length": 246.0, "rewards/meter/mean": 0.9970427751541138, "rewards/meter/std": 0.0011380411451682448, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.0739355981349945, "rewards/total_composite/mean": 0.7299137115478516, "rewards/total_composite/std": 0.07295885682106018, "reward": 0.7299137115478516, "reward_std": 0.07295886427164078, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016915393993258476, "sampling/sampling_logp_difference/max": 1.4178450107574463, "sampling/importance_sampling_ratio/min": 0.2422354817390442, "sampling/importance_sampling_ratio/mean": 1.000618815422058, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.061672994401305914, "clip_ratio/low_mean": 0.0077110049314796925, "clip_ratio/low_min": 0.0077110049314796925, "clip_ratio/high_mean": 0.006238285917788744, "clip_ratio/high_max": 0.006238285917788744, "clip_ratio/region_mean": 0.013949290849268436, "reward_total_mean": 0.7299137115478516, "reward_meter_mean": 0.9970427751541138, "reward_meter_std": 0.0011380411451682448, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.0739355981349945, "reward_total_composite_mean": 0.7299137115478516, "reward_total_composite_std": 0.07295885682106018} {"timestamp_utc": "2026-04-11T23:48:06Z", "mode": "train", "global_step": 1216, "epoch": 0.04884122585050408, "loss": -0.0023, "grad_norm": 1.560193657875061, "learning_rate": 6.318181818181819e-06, "num_tokens": 2747538.0, "completions/mean_length": 124.125, "completions/min_length": 122.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 124.125, "completions/min_terminated_length": 122.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.99362713098526, "rewards/meter/std": 0.0015980940079316497, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7678571343421936, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.7628991007804871, "rewards/total_composite/std": 0.07266794145107269, "reward": 0.7628991007804871, "reward_std": 0.07266796380281448, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010110512375831604, "sampling/sampling_logp_difference/max": 0.9531111717224121, "sampling/importance_sampling_ratio/min": 0.3855396807193756, "sampling/importance_sampling_ratio/mean": 0.9995880722999573, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03512992733158171, "clip_ratio/low_mean": 0.003024590201675892, "clip_ratio/low_min": 0.003024590201675892, "clip_ratio/high_mean": 0.006018434185534716, "clip_ratio/high_max": 0.006018434185534716, "clip_ratio/region_mean": 0.009043024387210608, "reward_total_mean": 0.7628991007804871, "reward_meter_mean": 0.99362713098526, "reward_meter_std": 0.0015980940079316497, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7678571343421936, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.7628991007804871, "reward_total_composite_std": 0.07266794145107269} {"timestamp_utc": "2026-04-11T23:48:12Z", "mode": "train", "global_step": 1217, "epoch": 0.04888139133228903, "loss": 0.0084, "grad_norm": 4.285740852355957, "learning_rate": 6.315151515151515e-06, "num_tokens": 2749384.0, "completions/mean_length": 75.75, "completions/min_length": 71.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.75, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.41745519638061523, "rewards/meter/std": 0.3040141463279724, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.3990457057952881, "rewards/total_composite/std": 0.3067740499973297, "reward": 0.3990457057952881, "reward_std": 0.3067740499973297, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03294277563691139, "sampling/sampling_logp_difference/max": 1.4083929061889648, "sampling/importance_sampling_ratio/min": 0.24453596770763397, "sampling/importance_sampling_ratio/mean": 1.002768635749817, "sampling/importance_sampling_ratio/max": 1.500490427017212, "entropy": 0.20184578280895948, "clip_ratio/low_mean": 0.009840250364504755, "clip_ratio/low_min": 0.009840250364504755, "clip_ratio/high_mean": 0.00835704104974866, "clip_ratio/high_max": 0.00835704104974866, "clip_ratio/region_mean": 0.018197291414253414, "reward_total_mean": 0.3990457057952881, "reward_meter_mean": 0.41745519638061523, "reward_meter_std": 0.3040141463279724, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.3990457057952881, "reward_total_composite_std": 0.3067740499973297} {"timestamp_utc": "2026-04-11T23:48:19Z", "mode": "train", "global_step": 1218, "epoch": 0.04892155681407399, "loss": -0.0029, "grad_norm": 4.59775972366333, "learning_rate": 6.3121212121212125e-06, "num_tokens": 2751115.0, "completions/mean_length": 64.375, "completions/min_length": 63.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.375, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9942485094070435, "rewards/meter/std": 0.003570063039660454, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9942485094070435, "rewards/total_composite/std": 0.003570063039660454, "reward": 0.9942485094070435, "reward_std": 0.0035700735170394182, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014056126587092876, "sampling/sampling_logp_difference/max": 1.6094756126403809, "sampling/importance_sampling_ratio/min": 0.19999246299266815, "sampling/importance_sampling_ratio/mean": 1.001212239265442, "sampling/importance_sampling_ratio/max": 1.7857288122177124, "entropy": 0.048549097031354904, "clip_ratio/low_mean": 0.003937252098694444, "clip_ratio/low_min": 0.003937252098694444, "clip_ratio/high_mean": 0.003968254197388887, "clip_ratio/high_max": 0.003968254197388887, "clip_ratio/region_mean": 0.007905506296083331, "reward_total_mean": 0.9942485094070435, "reward_meter_mean": 0.9942485094070435, "reward_meter_std": 0.003570063039660454, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9942485094070435, "reward_total_composite_std": 0.003570063039660454} {"timestamp_utc": "2026-04-11T23:48:25Z", "mode": "train", "global_step": 1219, "epoch": 0.04896172229585894, "loss": -0.0049, "grad_norm": 4.931517124176025, "learning_rate": 6.309090909090909e-06, "num_tokens": 2753014.0, "completions/mean_length": 78.375, "completions/min_length": 76.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.375, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9358141422271729, "rewards/meter/std": 0.024736735969781876, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9249999523162842, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.8663658499717712, "rewards/total_composite/std": 0.10636419802904129, "reward": 0.8663658499717712, "reward_std": 0.1063641905784607, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03355337679386139, "sampling/sampling_logp_difference/max": 2.9294357299804688, "sampling/importance_sampling_ratio/min": 0.05342717468738556, "sampling/importance_sampling_ratio/mean": 0.9951846599578857, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09925029519945383, "clip_ratio/low_mean": 0.008117978577502072, "clip_ratio/low_min": 0.008117978577502072, "clip_ratio/high_mean": 0.023660059785470366, "clip_ratio/high_max": 0.023660059785470366, "clip_ratio/region_mean": 0.03177803836297244, "reward_total_mean": 0.8663658499717712, "reward_meter_mean": 0.9358141422271729, "reward_meter_std": 0.024736735969781876, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9249999523162842, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.8663658499717712, "reward_total_composite_std": 0.10636419802904129} {"timestamp_utc": "2026-04-11T23:48:29Z", "mode": "train", "global_step": 1220, "epoch": 0.049001887777643895, "loss": 0.0002, "grad_norm": 0.8351545929908752, "learning_rate": 6.306060606060607e-06, "num_tokens": 2754558.0, "completions/mean_length": 37.0, "completions/min_length": 37.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.0, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9974673390388489, "rewards/meter/std": 1.9203745978302322e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9974673390388489, "rewards/total_composite/std": 1.9203745978302322e-05, "reward": 0.9974673390388489, "reward_std": 1.9203745978302322e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010439498350024223, "sampling/sampling_logp_difference/max": 0.9242507219314575, "sampling/importance_sampling_ratio/min": 0.39682865142822266, "sampling/importance_sampling_ratio/mean": 1.0021648406982422, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03344483021646738, "clip_ratio/low_mean": 0.0033783784601837397, "clip_ratio/low_min": 0.0033783784601837397, "clip_ratio/high_mean": 0.0033783784601837397, "clip_ratio/high_max": 0.0033783784601837397, "clip_ratio/region_mean": 0.006756756920367479, "reward_total_mean": 0.9974673390388489, "reward_meter_mean": 0.9974673390388489, "reward_meter_std": 1.9203745978302322e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9974673390388489, "reward_total_composite_std": 1.9203745978302322e-05} {"timestamp_utc": "2026-04-11T23:48:36Z", "mode": "train", "global_step": 1221, "epoch": 0.04904205325942885, "loss": 0.0928, "grad_norm": 3.4342517852783203, "learning_rate": 6.303030303030303e-06, "num_tokens": 2757419.0, "completions/mean_length": 174.625, "completions/min_length": 162.0, "completions/max_length": 219.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 174.625, "completions/min_terminated_length": 162.0, "completions/max_terminated_length": 219.0, "rewards/meter/mean": 0.9972714185714722, "rewards/meter/std": 0.0009564639185555279, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8042929172515869, "rewards/repeat_penalty/std": 0.05763205140829086, "rewards/total_composite/mean": 0.7635921239852905, "rewards/total_composite/std": 0.1048756018280983, "reward": 0.7635921239852905, "reward_std": 0.1048755943775177, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025953466072678566, "sampling/sampling_logp_difference/max": 4.84648323059082, "sampling/importance_sampling_ratio/min": 0.007855957373976707, "sampling/importance_sampling_ratio/mean": 0.9992591738700867, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06981830345466733, "clip_ratio/low_mean": 0.0034246575087308884, "clip_ratio/low_min": 0.0034246575087308884, "clip_ratio/high_mean": 0.00996978604234755, "clip_ratio/high_max": 0.00996978604234755, "clip_ratio/region_mean": 0.013394443551078439, "reward_total_mean": 0.7635921239852905, "reward_meter_mean": 0.9972714185714722, "reward_meter_std": 0.0009564639185555279, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8042929172515869, "reward_repeat_penalty_std": 0.05763205140829086, "reward_total_composite_mean": 0.7635921239852905, "reward_total_composite_std": 0.1048756018280983} {"timestamp_utc": "2026-04-11T23:48:44Z", "mode": "train", "global_step": 1222, "epoch": 0.0490822187412138, "loss": -0.0077, "grad_norm": 1.311091661453247, "learning_rate": 6.300000000000001e-06, "num_tokens": 2761573.0, "completions/mean_length": 276.25, "completions/min_length": 270.0, "completions/max_length": 286.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 276.25, "completions/min_terminated_length": 270.0, "completions/max_terminated_length": 286.0, "rewards/meter/mean": 0.9943939447402954, "rewards/meter/std": 0.001025979407131672, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5, "rewards/repeat_penalty/std": 0.15936382114887238, "rewards/total_composite/mean": 0.4972769021987915, "rewards/total_composite/std": 0.158652201294899, "reward": 0.4972769021987915, "reward_std": 0.1586521863937378, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013553553260862827, "sampling/sampling_logp_difference/max": 3.245562791824341, "sampling/importance_sampling_ratio/min": 0.03894663602113724, "sampling/importance_sampling_ratio/mean": 0.9976973533630371, "sampling/importance_sampling_ratio/max": 1.5534347295761108, "entropy": 0.04638162930496037, "clip_ratio/low_mean": 0.0027206970553379506, "clip_ratio/low_min": 0.0027206970553379506, "clip_ratio/high_mean": 0.006255614600377157, "clip_ratio/high_max": 0.006255614600377157, "clip_ratio/region_mean": 0.008976311655715108, "reward_total_mean": 0.4972769021987915, "reward_meter_mean": 0.9943939447402954, "reward_meter_std": 0.001025979407131672, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5, "reward_repeat_penalty_std": 0.15936382114887238, "reward_total_composite_mean": 0.4972769021987915, "reward_total_composite_std": 0.158652201294899} {"timestamp_utc": "2026-04-11T23:48:49Z", "mode": "train", "global_step": 1223, "epoch": 0.049122384222998756, "loss": -0.003, "grad_norm": 2.368471145629883, "learning_rate": 6.296969696969697e-06, "num_tokens": 2763119.0, "completions/mean_length": 34.25, "completions/min_length": 34.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.25, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9942933320999146, "rewards/meter/std": 8.478895324515179e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9942933320999146, "rewards/total_composite/std": 8.478895324515179e-05, "reward": 0.9942933320999146, "reward_std": 8.477975643472746e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009194603189826012, "sampling/sampling_logp_difference/max": 0.555586576461792, "sampling/importance_sampling_ratio/min": 0.5737356543540955, "sampling/importance_sampling_ratio/mean": 1.0032249689102173, "sampling/importance_sampling_ratio/max": 1.5068020820617676, "entropy": 0.043639433570206165, "clip_ratio/low_mean": 0.0036764706019312143, "clip_ratio/low_min": 0.0036764706019312143, "clip_ratio/high_mean": 0.0035714285913854837, "clip_ratio/high_max": 0.0035714285913854837, "clip_ratio/region_mean": 0.007247899193316698, "reward_total_mean": 0.9942933320999146, "reward_meter_mean": 0.9942933320999146, "reward_meter_std": 8.478895324515179e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9942933320999146, "reward_total_composite_std": 8.478895324515179e-05} {"timestamp_utc": "2026-04-11T23:48:54Z", "mode": "train", "global_step": 1224, "epoch": 0.04916254970478371, "loss": 0.0036, "grad_norm": 2.613860607147217, "learning_rate": 6.293939393939394e-06, "num_tokens": 2764882.0, "completions/mean_length": 75.375, "completions/min_length": 72.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.375, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9953185319900513, "rewards/meter/std": 0.0019400938181206584, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.8708760142326355, "rewards/total_composite/std": 0.17157325148582458, "reward": 0.8708760142326355, "reward_std": 0.1715732365846634, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028062716126441956, "sampling/sampling_logp_difference/max": 2.739466667175293, "sampling/importance_sampling_ratio/min": 0.06460478901863098, "sampling/importance_sampling_ratio/mean": 1.003928542137146, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.141799321398139, "clip_ratio/low_mean": 0.006672008661553264, "clip_ratio/low_min": 0.006672008661553264, "clip_ratio/high_mean": 0.016601469949819148, "clip_ratio/high_max": 0.016601469949819148, "clip_ratio/region_mean": 0.02327347861137241, "reward_total_mean": 0.8708760142326355, "reward_meter_mean": 0.9953185319900513, "reward_meter_std": 0.0019400938181206584, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.17251639068126678, "reward_total_composite_mean": 0.8708760142326355, "reward_total_composite_std": 0.17157325148582458} {"timestamp_utc": "2026-04-11T23:49:00Z", "mode": "train", "global_step": 1225, "epoch": 0.049202715186568664, "loss": -0.0013, "grad_norm": 0.7861045002937317, "learning_rate": 6.290909090909092e-06, "num_tokens": 2767323.0, "completions/mean_length": 122.125, "completions/min_length": 122.0, "completions/max_length": 123.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.125, "completions/min_terminated_length": 122.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.9720443487167358, "rewards/meter/std": 0.0028714225627481937, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.2857142984867096, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.277726948261261, "rewards/total_composite/std": 0.000820409506559372, "reward": 0.277726948261261, "reward_std": 0.0008204064215533435, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009138394147157669, "sampling/sampling_logp_difference/max": 2.7687785625457764, "sampling/importance_sampling_ratio/min": 0.12622418999671936, "sampling/importance_sampling_ratio/mean": 0.9990015625953674, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.018890328239649534, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0030487803742289543, "clip_ratio/high_max": 0.0030487803742289543, "clip_ratio/region_mean": 0.0030487803742289543, "reward_total_mean": 0.277726948261261, "reward_meter_mean": 0.9720443487167358, "reward_meter_std": 0.0028714225627481937, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.2857142984867096, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.277726948261261, "reward_total_composite_std": 0.000820409506559372} {"timestamp_utc": "2026-04-11T23:49:05Z", "mode": "train", "global_step": 1226, "epoch": 0.04924288066835362, "loss": -0.0158, "grad_norm": 7.526054382324219, "learning_rate": 6.287878787878788e-06, "num_tokens": 2768989.0, "completions/mean_length": 62.25, "completions/min_length": 58.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.25, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.8702743053436279, "rewards/meter/std": 0.2847423553466797, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.75, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.6277604103088379, "rewards/total_composite/std": 0.21942587196826935, "reward": 0.6277604103088379, "reward_std": 0.21942584216594696, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027990855276584625, "sampling/sampling_logp_difference/max": 1.2952322959899902, "sampling/importance_sampling_ratio/min": 0.273834228515625, "sampling/importance_sampling_ratio/mean": 1.007242202758789, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10299600008875132, "clip_ratio/low_mean": 0.004310344811528921, "clip_ratio/low_min": 0.004310344811528921, "clip_ratio/high_mean": 0.020103482995182276, "clip_ratio/high_max": 0.020103482995182276, "clip_ratio/region_mean": 0.024413827806711197, "reward_total_mean": 0.6277604103088379, "reward_meter_mean": 0.8702743053436279, "reward_meter_std": 0.2847423553466797, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.75, "reward_repeat_penalty_std": 0.15430334210395813, "reward_total_composite_mean": 0.6277604103088379, "reward_total_composite_std": 0.21942587196826935} {"timestamp_utc": "2026-04-11T23:49:09Z", "mode": "train", "global_step": 1227, "epoch": 0.04928304615013857, "loss": -0.0048, "grad_norm": 1.5801409482955933, "learning_rate": 6.284848484848486e-06, "num_tokens": 2770476.0, "completions/mean_length": 32.875, "completions/min_length": 32.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.875, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9896509051322937, "rewards/meter/std": 0.00014116229431238025, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9896509051322937, "rewards/total_composite/std": 0.00014116229431238025, "reward": 0.9896509051322937, "reward_std": 0.00014115417434368283, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016333717852830887, "sampling/sampling_logp_difference/max": 1.630277156829834, "sampling/importance_sampling_ratio/min": 0.19587527215480804, "sampling/importance_sampling_ratio/mean": 0.9984188675880432, "sampling/importance_sampling_ratio/max": 1.3669874668121338, "entropy": 0.03647553990595043, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.00390625, "reward_total_mean": 0.9896509051322937, "reward_meter_mean": 0.9896509051322937, "reward_meter_std": 0.00014116229431238025, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9896509051322937, "reward_total_composite_std": 0.00014116229431238025} {"timestamp_utc": "2026-04-11T23:49:15Z", "mode": "train", "global_step": 1228, "epoch": 0.049323211631923526, "loss": 0.0107, "grad_norm": 3.883788824081421, "learning_rate": 6.2818181818181825e-06, "num_tokens": 2772825.0, "completions/mean_length": 131.625, "completions/min_length": 126.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.625, "completions/min_terminated_length": 126.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.9974247217178345, "rewards/meter/std": 0.001207518856972456, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8035714626312256, "rewards/repeat_penalty/std": 0.10628911107778549, "rewards/total_composite/mean": 0.8014565706253052, "rewards/total_composite/std": 0.10568944364786148, "reward": 0.8014565706253052, "reward_std": 0.10568942874670029, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01869257725775242, "sampling/sampling_logp_difference/max": 3.0793256759643555, "sampling/importance_sampling_ratio/min": 0.04599026218056679, "sampling/importance_sampling_ratio/mean": 1.0004632472991943, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0591668093111366, "clip_ratio/low_mean": 0.004665242100600153, "clip_ratio/low_min": 0.004665242100600153, "clip_ratio/high_mean": 0.009555137949064374, "clip_ratio/high_max": 0.009555137949064374, "clip_ratio/region_mean": 0.014220380049664527, "reward_total_mean": 0.8014565706253052, "reward_meter_mean": 0.9974247217178345, "reward_meter_std": 0.001207518856972456, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8035714626312256, "reward_repeat_penalty_std": 0.10628911107778549, "reward_total_composite_mean": 0.8014565706253052, "reward_total_composite_std": 0.10568944364786148} {"timestamp_utc": "2026-04-11T23:49:20Z", "mode": "train", "global_step": 1229, "epoch": 0.04936337711370848, "loss": 0.0111, "grad_norm": 6.317329406738281, "learning_rate": 6.27878787878788e-06, "num_tokens": 2774629.0, "completions/mean_length": 57.5, "completions/min_length": 56.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.5, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.11471378803253174, "rewards/meter/std": 0.10604586452245712, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.11471378803253174, "rewards/total_composite/std": 0.10604586452245712, "reward": 0.11471378803253174, "reward_std": 0.10604586452245712, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.033396147191524506, "sampling/sampling_logp_difference/max": 1.2361993789672852, "sampling/importance_sampling_ratio/min": 0.2904861271381378, "sampling/importance_sampling_ratio/mean": 1.000756025314331, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15489636361598969, "clip_ratio/low_mean": 0.006359649356454611, "clip_ratio/low_min": 0.006359649356454611, "clip_ratio/high_mean": 0.008696309756487608, "clip_ratio/high_max": 0.008696309756487608, "clip_ratio/region_mean": 0.015055959112942219, "reward_total_mean": 0.11471378803253174, "reward_meter_mean": 0.11471378803253174, "reward_meter_std": 0.10604586452245712, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.11471378803253174, "reward_total_composite_std": 0.10604586452245712} {"timestamp_utc": "2026-04-11T23:49:24Z", "mode": "train", "global_step": 1230, "epoch": 0.049403542595493434, "loss": 0.0035, "grad_norm": 3.3790931701660156, "learning_rate": 6.275757575757576e-06, "num_tokens": 2776041.0, "completions/mean_length": 33.5, "completions/min_length": 33.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.5, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9702644944190979, "rewards/meter/std": 0.0008763103978708386, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9702644944190979, "rewards/total_composite/std": 0.0008763103978708386, "reward": 0.9702644944190979, "reward_std": 0.0008763210498727858, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022185776382684708, "sampling/sampling_logp_difference/max": 1.2655248641967773, "sampling/importance_sampling_ratio/min": 0.3646947145462036, "sampling/importance_sampling_ratio/mean": 1.0119966268539429, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07495713606476784, "clip_ratio/low_mean": 0.007464349502697587, "clip_ratio/low_min": 0.007464349502697587, "clip_ratio/high_mean": 0.0036764706019312143, "clip_ratio/high_max": 0.0036764706019312143, "clip_ratio/region_mean": 0.011140820104628801, "reward_total_mean": 0.9702644944190979, "reward_meter_mean": 0.9702644944190979, "reward_meter_std": 0.0008763103978708386, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9702644944190979, "reward_total_composite_std": 0.0008763103978708386} {"timestamp_utc": "2026-04-11T23:49:29Z", "mode": "train", "global_step": 1231, "epoch": 0.04944370807727839, "loss": -0.0146, "grad_norm": 6.076712608337402, "learning_rate": 6.2727272727272734e-06, "num_tokens": 2777725.0, "completions/mean_length": 58.5, "completions/min_length": 52.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.5, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.6401726603507996, "rewards/meter/std": 0.34182804822921753, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6401726603507996, "rewards/total_composite/std": 0.34182804822921753, "reward": 0.6401726603507996, "reward_std": 0.34182801842689514, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06882074475288391, "sampling/sampling_logp_difference/max": 1.4318437576293945, "sampling/importance_sampling_ratio/min": 0.24655072391033173, "sampling/importance_sampling_ratio/mean": 1.0154814720153809, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5035125408321619, "clip_ratio/low_mean": 0.028201664797961712, "clip_ratio/low_min": 0.028201664797961712, "clip_ratio/high_mean": 0.03711274731904268, "clip_ratio/high_max": 0.03711274731904268, "clip_ratio/region_mean": 0.0653144121170044, "reward_total_mean": 0.6401726603507996, "reward_meter_mean": 0.6401726603507996, "reward_meter_std": 0.34182804822921753, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6401726603507996, "reward_total_composite_std": 0.34182804822921753} {"timestamp_utc": "2026-04-11T23:49:34Z", "mode": "train", "global_step": 1232, "epoch": 0.04948387355906334, "loss": -0.0004, "grad_norm": 4.830690860748291, "learning_rate": 6.26969696969697e-06, "num_tokens": 2779575.0, "completions/mean_length": 76.25, "completions/min_length": 74.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.25, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.8763631582260132, "rewards/meter/std": 0.33702343702316284, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8763631582260132, "rewards/total_composite/std": 0.33702343702316284, "reward": 0.8763631582260132, "reward_std": 0.33702343702316284, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03446757048368454, "sampling/sampling_logp_difference/max": 3.854651927947998, "sampling/importance_sampling_ratio/min": 0.021180974319577217, "sampling/importance_sampling_ratio/mean": 0.9978901743888855, "sampling/importance_sampling_ratio/max": 1.7376562356948853, "entropy": 0.1252383915707469, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.01965810963883996, "clip_ratio/high_max": 0.01965810963883996, "clip_ratio/region_mean": 0.01965810963883996, "reward_total_mean": 0.8763631582260132, "reward_meter_mean": 0.8763631582260132, "reward_meter_std": 0.33702343702316284, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8763631582260132, "reward_total_composite_std": 0.33702343702316284} {"timestamp_utc": "2026-04-11T23:49:41Z", "mode": "train", "global_step": 1233, "epoch": 0.049524039040848296, "loss": 0.0154, "grad_norm": 2.373234748840332, "learning_rate": 6.266666666666668e-06, "num_tokens": 2782216.0, "completions/mean_length": 150.125, "completions/min_length": 140.0, "completions/max_length": 152.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 150.125, "completions/min_terminated_length": 140.0, "completions/max_terminated_length": 152.0, "rewards/meter/mean": 0.9954185485839844, "rewards/meter/std": 0.0016214650822803378, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6785714626312256, "rewards/repeat_penalty/std": 0.14787118136882782, "rewards/total_composite/mean": 0.6755450963973999, "rewards/total_composite/std": 0.1476629376411438, "reward": 0.6755450963973999, "reward_std": 0.1476629078388214, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013783158734440804, "sampling/sampling_logp_difference/max": 1.0632364749908447, "sampling/importance_sampling_ratio/min": 0.34533634781837463, "sampling/importance_sampling_ratio/mean": 1.002487301826477, "sampling/importance_sampling_ratio/max": 1.8110910654067993, "entropy": 0.08290160167962313, "clip_ratio/low_mean": 0.004945102846249938, "clip_ratio/low_min": 0.004945102846249938, "clip_ratio/high_mean": 0.006812343490310013, "clip_ratio/high_max": 0.006812343490310013, "clip_ratio/region_mean": 0.011757446336559951, "reward_total_mean": 0.6755450963973999, "reward_meter_mean": 0.9954185485839844, "reward_meter_std": 0.0016214650822803378, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6785714626312256, "reward_repeat_penalty_std": 0.14787118136882782, "reward_total_composite_mean": 0.6755450963973999, "reward_total_composite_std": 0.1476629376411438} {"timestamp_utc": "2026-04-11T23:49:48Z", "mode": "train", "global_step": 1234, "epoch": 0.04956420452263325, "loss": -0.002, "grad_norm": 1.7159820795059204, "learning_rate": 6.263636363636364e-06, "num_tokens": 2785815.0, "completions/mean_length": 242.875, "completions/min_length": 227.0, "completions/max_length": 260.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 242.875, "completions/min_terminated_length": 227.0, "completions/max_terminated_length": 260.0, "rewards/meter/mean": 0.9932608008384705, "rewards/meter/std": 0.003682313719764352, "rewards/count_adherence/mean": 0.737500011920929, "rewards/count_adherence/std": 0.05175492912530899, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6418956518173218, "rewards/repeat_penalty/std": 0.06913584470748901, "rewards/total_composite/mean": 0.4686053991317749, "rewards/total_composite/std": 0.04152140021324158, "reward": 0.4686053991317749, "reward_std": 0.04152139648795128, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022554513067007065, "sampling/sampling_logp_difference/max": 3.497123956680298, "sampling/importance_sampling_ratio/min": 0.030284356325864792, "sampling/importance_sampling_ratio/mean": 1.0014477968215942, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09457159182056785, "clip_ratio/low_mean": 0.005789120797999203, "clip_ratio/low_min": 0.005789120797999203, "clip_ratio/high_mean": 0.00862484163371846, "clip_ratio/high_max": 0.00862484163371846, "clip_ratio/region_mean": 0.014413962431717664, "reward_total_mean": 0.4686053991317749, "reward_meter_mean": 0.9932608008384705, "reward_meter_std": 0.003682313719764352, "reward_count_adherence_mean": 0.737500011920929, "reward_count_adherence_std": 0.05175492912530899, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6418956518173218, "reward_repeat_penalty_std": 0.06913584470748901, "reward_total_composite_mean": 0.4686053991317749, "reward_total_composite_std": 0.04152140021324158} {"timestamp_utc": "2026-04-11T23:49:53Z", "mode": "train", "global_step": 1235, "epoch": 0.049604370004418204, "loss": 0.011, "grad_norm": 15.415648460388184, "learning_rate": 6.260606060606062e-06, "num_tokens": 2787651.0, "completions/mean_length": 80.5, "completions/min_length": 79.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.5, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.9555257558822632, "rewards/meter/std": 0.013856525532901287, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9315690398216248, "rewards/total_composite/std": 0.06806813180446625, "reward": 0.9315690398216248, "reward_std": 0.06806813925504684, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04006582126021385, "sampling/sampling_logp_difference/max": 5.584328651428223, "sampling/importance_sampling_ratio/min": 0.003756270743906498, "sampling/importance_sampling_ratio/mean": 0.9921613335609436, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10135750938206911, "clip_ratio/low_mean": 0.006172839552164078, "clip_ratio/low_min": 0.006172839552164078, "clip_ratio/high_mean": 0.031061405315995216, "clip_ratio/high_max": 0.031061405315995216, "clip_ratio/region_mean": 0.037234244868159294, "reward_total_mean": 0.9315690398216248, "reward_meter_mean": 0.9555257558822632, "reward_meter_std": 0.013856525532901287, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9315690398216248, "reward_total_composite_std": 0.06806813180446625} {"timestamp_utc": "2026-04-11T23:49:59Z", "mode": "train", "global_step": 1236, "epoch": 0.04964453548620316, "loss": 0.0094, "grad_norm": 1.9874340295791626, "learning_rate": 6.257575757575758e-06, "num_tokens": 2790696.0, "completions/mean_length": 181.625, "completions/min_length": 173.0, "completions/max_length": 186.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 181.625, "completions/min_terminated_length": 173.0, "completions/max_terminated_length": 186.0, "rewards/meter/mean": 0.9979325532913208, "rewards/meter/std": 0.00011955534137086943, "rewards/count_adherence/mean": 0.7142857313156128, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.737500011920929, "rewards/repeat_penalty/std": 0.07440238445997238, "rewards/total_composite/mean": 0.5256986618041992, "rewards/total_composite/std": 0.053062621504068375, "reward": 0.5256986618041992, "reward_std": 0.05306261032819748, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012886044569313526, "sampling/sampling_logp_difference/max": 1.4943835735321045, "sampling/importance_sampling_ratio/min": 0.22438688576221466, "sampling/importance_sampling_ratio/mean": 1.0016508102416992, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.050744547275826335, "clip_ratio/low_mean": 0.002725058700889349, "clip_ratio/low_min": 0.002725058700889349, "clip_ratio/high_mean": 0.006200521253049374, "clip_ratio/high_max": 0.006200521253049374, "clip_ratio/region_mean": 0.008925579953938723, "reward_total_mean": 0.5256986618041992, "reward_meter_mean": 0.9979325532913208, "reward_meter_std": 0.00011955534137086943, "reward_count_adherence_mean": 0.7142857313156128, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.737500011920929, "reward_repeat_penalty_std": 0.07440238445997238, "reward_total_composite_mean": 0.5256986618041992, "reward_total_composite_std": 0.053062621504068375} {"timestamp_utc": "2026-04-11T23:50:04Z", "mode": "train", "global_step": 1237, "epoch": 0.04968470096798811, "loss": 0.0102, "grad_norm": 6.145448684692383, "learning_rate": 6.254545454545455e-06, "num_tokens": 2792675.0, "completions/mean_length": 67.375, "completions/min_length": 64.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.375, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.35231250524520874, "rewards/meter/std": 0.3081033229827881, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.35225385427474976, "rewards/total_composite/std": 0.30817967653274536, "reward": 0.35225385427474976, "reward_std": 0.30817967653274536, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0802878737449646, "sampling/sampling_logp_difference/max": 3.271935224533081, "sampling/importance_sampling_ratio/min": 0.037932947278022766, "sampling/importance_sampling_ratio/mean": 1.002850890159607, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3711269237101078, "clip_ratio/low_mean": 0.03385214158333838, "clip_ratio/low_min": 0.03385214158333838, "clip_ratio/high_mean": 0.028940699994564056, "clip_ratio/high_max": 0.028940699994564056, "clip_ratio/region_mean": 0.06279284157790244, "reward_total_mean": 0.35225385427474976, "reward_meter_mean": 0.35231250524520874, "reward_meter_std": 0.3081033229827881, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.35225385427474976, "reward_total_composite_std": 0.30817967653274536} {"timestamp_utc": "2026-04-11T23:50:09Z", "mode": "train", "global_step": 1238, "epoch": 0.049724866449773066, "loss": -0.0256, "grad_norm": 6.303664684295654, "learning_rate": 6.251515151515152e-06, "num_tokens": 2794768.0, "completions/mean_length": 85.625, "completions/min_length": 78.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.625, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.6144058704376221, "rewards/meter/std": 0.3429763913154602, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.15430334210395813, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.452033668756485, "rewards/total_composite/std": 0.2605564594268799, "reward": 0.452033668756485, "reward_std": 0.2605564594268799, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05436992645263672, "sampling/sampling_logp_difference/max": 2.363379955291748, "sampling/importance_sampling_ratio/min": 0.09410162270069122, "sampling/importance_sampling_ratio/mean": 1.0040868520736694, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1898734886199236, "clip_ratio/low_mean": 0.009122495306655765, "clip_ratio/low_min": 0.009122495306655765, "clip_ratio/high_mean": 0.017562699620611966, "clip_ratio/high_max": 0.017562699620611966, "clip_ratio/region_mean": 0.02668519492726773, "reward_total_mean": 0.452033668756485, "reward_meter_mean": 0.6144058704376221, "reward_meter_std": 0.3429763913154602, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.15430334210395813, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.452033668756485, "reward_total_composite_std": 0.2605564594268799} {"timestamp_utc": "2026-04-11T23:50:14Z", "mode": "train", "global_step": 1239, "epoch": 0.04976503193155802, "loss": 0.0011, "grad_norm": 11.653518676757812, "learning_rate": 6.248484848484849e-06, "num_tokens": 2796176.0, "completions/mean_length": 34.0, "completions/min_length": 33.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9623700380325317, "rewards/meter/std": 0.017837192863225937, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9623700380325317, "rewards/total_composite/std": 0.017837192863225937, "reward": 0.9623700380325317, "reward_std": 0.017837194725871086, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017896274104714394, "sampling/sampling_logp_difference/max": 1.5471727848052979, "sampling/importance_sampling_ratio/min": 0.21284890174865723, "sampling/importance_sampling_ratio/mean": 0.9969296455383301, "sampling/importance_sampling_ratio/max": 1.2581915855407715, "entropy": 0.04970884299837053, "clip_ratio/low_mean": 0.0036764706019312143, "clip_ratio/low_min": 0.0036764706019312143, "clip_ratio/high_mean": 0.0037878789007663727, "clip_ratio/high_max": 0.0037878789007663727, "clip_ratio/region_mean": 0.007464349502697587, "reward_total_mean": 0.9623700380325317, "reward_meter_mean": 0.9623700380325317, "reward_meter_std": 0.017837192863225937, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9623700380325317, "reward_total_composite_std": 0.017837192863225937} {"timestamp_utc": "2026-04-11T23:50:19Z", "mode": "train", "global_step": 1240, "epoch": 0.04980519741334297, "loss": -0.0239, "grad_norm": 18.420658111572266, "learning_rate": 6.245454545454545e-06, "num_tokens": 2797992.0, "completions/mean_length": 65.0, "completions/min_length": 58.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.0, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.750508189201355, "rewards/meter/std": 0.30968019366264343, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.750508189201355, "rewards/total_composite/std": 0.30968019366264343, "reward": 0.750508189201355, "reward_std": 0.3096802234649658, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.052898943424224854, "sampling/sampling_logp_difference/max": 2.5542707443237305, "sampling/importance_sampling_ratio/min": 0.07774890959262848, "sampling/importance_sampling_ratio/mean": 0.9967520833015442, "sampling/importance_sampling_ratio/max": 1.6765276193618774, "entropy": 0.17542924359440804, "clip_ratio/low_mean": 0.023755351547151804, "clip_ratio/low_min": 0.023755351547151804, "clip_ratio/high_mean": 0.037521140300668776, "clip_ratio/high_max": 0.037521140300668776, "clip_ratio/region_mean": 0.06127649184782058, "reward_total_mean": 0.750508189201355, "reward_meter_mean": 0.750508189201355, "reward_meter_std": 0.30968019366264343, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.750508189201355, "reward_total_composite_std": 0.30968019366264343} {"timestamp_utc": "2026-04-11T23:50:24Z", "mode": "train", "global_step": 1241, "epoch": 0.04984536289512793, "loss": 0.0106, "grad_norm": 5.958988189697266, "learning_rate": 6.2424242424242434e-06, "num_tokens": 2799886.0, "completions/mean_length": 69.75, "completions/min_length": 67.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.75, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9957868456840515, "rewards/meter/std": 0.0033736079931259155, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957868456840515, "rewards/total_composite/std": 0.0033736079931259155, "reward": 0.9957868456840515, "reward_std": 0.0033736105542629957, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0355050228536129, "sampling/sampling_logp_difference/max": 1.2876784801483154, "sampling/importance_sampling_ratio/min": 0.2759105861186981, "sampling/importance_sampling_ratio/mean": 0.9955186247825623, "sampling/importance_sampling_ratio/max": 1.7876514196395874, "entropy": 0.134712896309793, "clip_ratio/low_mean": 0.007147361640818417, "clip_ratio/low_min": 0.007147361640818417, "clip_ratio/high_mean": 0.030546388239599764, "clip_ratio/high_max": 0.030546388239599764, "clip_ratio/region_mean": 0.03769374988041818, "reward_total_mean": 0.9957868456840515, "reward_meter_mean": 0.9957868456840515, "reward_meter_std": 0.0033736079931259155, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9957868456840515, "reward_total_composite_std": 0.0033736079931259155} {"timestamp_utc": "2026-04-11T23:50:29Z", "mode": "train", "global_step": 1242, "epoch": 0.04988552837691288, "loss": 0.0265, "grad_norm": 16.907014846801758, "learning_rate": 6.23939393939394e-06, "num_tokens": 2801684.0, "completions/mean_length": 70.75, "completions/min_length": 68.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.75, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9978107810020447, "rewards/meter/std": 0.0009332987247034907, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978107810020447, "rewards/total_composite/std": 0.0009332987247034907, "reward": 0.9978107810020447, "reward_std": 0.0009333047200925648, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03522227331995964, "sampling/sampling_logp_difference/max": 2.570969581604004, "sampling/importance_sampling_ratio/min": 0.07646137475967407, "sampling/importance_sampling_ratio/mean": 1.003254771232605, "sampling/importance_sampling_ratio/max": 1.5690217018127441, "entropy": 0.14225120842456818, "clip_ratio/low_mean": 0.006905802641995251, "clip_ratio/low_min": 0.006905802641995251, "clip_ratio/high_mean": 0.024632065324112773, "clip_ratio/high_max": 0.024632065324112773, "clip_ratio/region_mean": 0.031537867966108024, "reward_total_mean": 0.9978107810020447, "reward_meter_mean": 0.9978107810020447, "reward_meter_std": 0.0009332987247034907, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9978107810020447, "reward_total_composite_std": 0.0009332987247034907} {"timestamp_utc": "2026-04-11T23:50:34Z", "mode": "train", "global_step": 1243, "epoch": 0.049925693858697835, "loss": 0.0058, "grad_norm": 8.487194061279297, "learning_rate": 6.236363636363637e-06, "num_tokens": 2803443.0, "completions/mean_length": 63.875, "completions/min_length": 63.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.875, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9128004312515259, "rewards/meter/std": 0.03738430514931679, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9128004312515259, "rewards/total_composite/std": 0.03738430514931679, "reward": 0.9128004312515259, "reward_std": 0.037384290248155594, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04461809620261192, "sampling/sampling_logp_difference/max": 5.818354606628418, "sampling/importance_sampling_ratio/min": 0.002972492016851902, "sampling/importance_sampling_ratio/mean": 1.0069886445999146, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10157017782330513, "clip_ratio/low_mean": 0.021577381063252687, "clip_ratio/low_min": 0.021577381063252687, "clip_ratio/high_mean": 0.017581941094249487, "clip_ratio/high_max": 0.017581941094249487, "clip_ratio/region_mean": 0.039159322157502174, "reward_total_mean": 0.9128004312515259, "reward_meter_mean": 0.9128004312515259, "reward_meter_std": 0.03738430514931679, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9128004312515259, "reward_total_composite_std": 0.03738430514931679} {"timestamp_utc": "2026-04-11T23:50:40Z", "mode": "train", "global_step": 1244, "epoch": 0.04996585934048279, "loss": -0.0386, "grad_norm": 2.8616859912872314, "learning_rate": 6.2333333333333335e-06, "num_tokens": 2805756.0, "completions/mean_length": 128.125, "completions/min_length": 110.0, "completions/max_length": 134.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.125, "completions/min_terminated_length": 110.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.9177215695381165, "rewards/meter/std": 0.08298756927251816, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8363094925880432, "rewards/repeat_penalty/std": 0.09127599745988846, "rewards/total_composite/mean": 0.7401801347732544, "rewards/total_composite/std": 0.10087206214666367, "reward": 0.7401801347732544, "reward_std": 0.10087206959724426, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021584779024124146, "sampling/sampling_logp_difference/max": 0.7692317962646484, "sampling/importance_sampling_ratio/min": 0.46336889266967773, "sampling/importance_sampling_ratio/mean": 1.0052604675292969, "sampling/importance_sampling_ratio/max": 1.7540781497955322, "entropy": 0.1307503404095769, "clip_ratio/low_mean": 0.009928334911819547, "clip_ratio/low_min": 0.009928334911819547, "clip_ratio/high_mean": 0.009529390663374215, "clip_ratio/high_max": 0.009529390663374215, "clip_ratio/region_mean": 0.019457725575193763, "reward_total_mean": 0.7401801347732544, "reward_meter_mean": 0.9177215695381165, "reward_meter_std": 0.08298756927251816, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8363094925880432, "reward_repeat_penalty_std": 0.09127599745988846, "reward_total_composite_mean": 0.7401801347732544, "reward_total_composite_std": 0.10087206214666367} {"timestamp_utc": "2026-04-11T23:50:47Z", "mode": "train", "global_step": 1245, "epoch": 0.05000602482226774, "loss": 0.0098, "grad_norm": 2.1829257011413574, "learning_rate": 6.230303030303031e-06, "num_tokens": 2809292.0, "completions/mean_length": 230.0, "completions/min_length": 218.0, "completions/max_length": 239.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 230.0, "completions/min_terminated_length": 218.0, "completions/max_terminated_length": 239.0, "rewards/meter/mean": 0.9974883198738098, "rewards/meter/std": 0.0011670741951093078, "rewards/count_adherence/mean": 0.7638888955116272, "rewards/count_adherence/std": 0.03928370773792267, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7868589758872986, "rewards/repeat_penalty/std": 0.07841575890779495, "rewards/total_composite/mean": 0.5989092588424683, "rewards/total_composite/std": 0.061690803617239, "reward": 0.5989092588424683, "reward_std": 0.061690811067819595, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01827998273074627, "sampling/sampling_logp_difference/max": 2.269944667816162, "sampling/importance_sampling_ratio/min": 0.10331789404153824, "sampling/importance_sampling_ratio/mean": 0.9981163740158081, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06460519693791866, "clip_ratio/low_mean": 0.009747638367116451, "clip_ratio/low_min": 0.009747638367116451, "clip_ratio/high_mean": 0.004898735671304166, "clip_ratio/high_max": 0.004898735671304166, "clip_ratio/region_mean": 0.014646374038420618, "reward_total_mean": 0.5989092588424683, "reward_meter_mean": 0.9974883198738098, "reward_meter_std": 0.0011670741951093078, "reward_count_adherence_mean": 0.7638888955116272, "reward_count_adherence_std": 0.03928370773792267, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7868589758872986, "reward_repeat_penalty_std": 0.07841575890779495, "reward_total_composite_mean": 0.5989092588424683, "reward_total_composite_std": 0.061690803617239} {"timestamp_utc": "2026-04-11T23:50:52Z", "mode": "train", "global_step": 1246, "epoch": 0.0500461903040527, "loss": -0.0011, "grad_norm": 5.752688407897949, "learning_rate": 6.227272727272727e-06, "num_tokens": 2811283.0, "completions/mean_length": 68.875, "completions/min_length": 68.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.875, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.7574069499969482, "rewards/meter/std": 0.31513679027557373, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7574069499969482, "rewards/total_composite/std": 0.31513679027557373, "reward": 0.7574069499969482, "reward_std": 0.31513676047325134, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017493808642029762, "sampling/sampling_logp_difference/max": 0.742142915725708, "sampling/importance_sampling_ratio/min": 0.4760926067829132, "sampling/importance_sampling_ratio/mean": 0.9997113347053528, "sampling/importance_sampling_ratio/max": 1.6327474117279053, "entropy": 0.08360549993813038, "clip_ratio/low_mean": 0.007273018010891974, "clip_ratio/low_min": 0.007273018010891974, "clip_ratio/high_mean": 0.007221258128993213, "clip_ratio/high_max": 0.007221258128993213, "clip_ratio/region_mean": 0.014494276139885187, "reward_total_mean": 0.7574069499969482, "reward_meter_mean": 0.7574069499969482, "reward_meter_std": 0.31513679027557373, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7574069499969482, "reward_total_composite_std": 0.31513679027557373} {"timestamp_utc": "2026-04-11T23:50:57Z", "mode": "train", "global_step": 1247, "epoch": 0.05008635578583765, "loss": -0.0245, "grad_norm": 3.6555347442626953, "learning_rate": 6.224242424242425e-06, "num_tokens": 2813271.0, "completions/mean_length": 95.5, "completions/min_length": 82.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.5, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.95280921459198, "rewards/meter/std": 0.017720038071274757, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.8177590370178223, "rewards/total_composite/std": 0.12049128115177155, "reward": 0.8177590370178223, "reward_std": 0.12049128115177155, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020115824416279793, "sampling/sampling_logp_difference/max": 0.5828499794006348, "sampling/importance_sampling_ratio/min": 0.5583049654960632, "sampling/importance_sampling_ratio/mean": 1.0025370121002197, "sampling/importance_sampling_ratio/max": 1.4265241622924805, "entropy": 0.13955960143357515, "clip_ratio/low_mean": 0.011977351736277342, "clip_ratio/low_min": 0.011977351736277342, "clip_ratio/high_mean": 0.007731958641670644, "clip_ratio/high_max": 0.007731958641670644, "clip_ratio/region_mean": 0.019709310377947986, "reward_total_mean": 0.8177590370178223, "reward_meter_mean": 0.95280921459198, "reward_meter_std": 0.017720038071274757, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_total_composite_mean": 0.8177590370178223, "reward_total_composite_std": 0.12049128115177155} {"timestamp_utc": "2026-04-11T23:51:02Z", "mode": "train", "global_step": 1248, "epoch": 0.050126521267622605, "loss": 0.0024, "grad_norm": 3.7371089458465576, "learning_rate": 6.221212121212121e-06, "num_tokens": 2815189.0, "completions/mean_length": 76.75, "completions/min_length": 75.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.75, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9958518147468567, "rewards/meter/std": 0.0027126488275825977, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9958518147468567, "rewards/total_composite/std": 0.0027126488275825977, "reward": 0.9958518147468567, "reward_std": 0.0027126302011311054, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03340546786785126, "sampling/sampling_logp_difference/max": 3.952702045440674, "sampling/importance_sampling_ratio/min": 0.019202744588255882, "sampling/importance_sampling_ratio/mean": 0.999036431312561, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.14584199711680412, "clip_ratio/low_mean": 0.004895833437331021, "clip_ratio/low_min": 0.004895833437331021, "clip_ratio/high_mean": 0.009827935369685292, "clip_ratio/high_max": 0.009827935369685292, "clip_ratio/region_mean": 0.014723768807016313, "reward_total_mean": 0.9958518147468567, "reward_meter_mean": 0.9958518147468567, "reward_meter_std": 0.0027126488275825977, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9958518147468567, "reward_total_composite_std": 0.0027126488275825977} {"timestamp_utc": "2026-04-11T23:51:07Z", "mode": "train", "global_step": 1249, "epoch": 0.05016668674940756, "loss": -0.0042, "grad_norm": 2.240663528442383, "learning_rate": 6.218181818181819e-06, "num_tokens": 2817451.0, "completions/mean_length": 100.75, "completions/min_length": 99.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.75, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.9981772899627686, "rewards/meter/std": 0.0003791229974012822, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.898347020149231, "rewards/total_composite/std": 0.1065896674990654, "reward": 0.898347020149231, "reward_std": 0.1065896600484848, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012546716257929802, "sampling/sampling_logp_difference/max": 0.8134353160858154, "sampling/importance_sampling_ratio/min": 0.516797661781311, "sampling/importance_sampling_ratio/mean": 1.003831148147583, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06938632484525442, "clip_ratio/low_mean": 0.007488123839721084, "clip_ratio/low_min": 0.007488123839721084, "clip_ratio/high_mean": 0.0036526747280731797, "clip_ratio/high_max": 0.0036526747280731797, "clip_ratio/region_mean": 0.011140798567794263, "reward_total_mean": 0.898347020149231, "reward_meter_mean": 0.9981772899627686, "reward_meter_std": 0.0003791229974012822, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_total_composite_mean": 0.898347020149231, "reward_total_composite_std": 0.1065896674990654} {"timestamp_utc": "2026-04-11T23:51:12Z", "mode": "train", "global_step": 1250, "epoch": 0.05020685223119251, "loss": -0.0122, "grad_norm": 2.518706798553467, "learning_rate": 6.215151515151515e-06, "num_tokens": 2819322.0, "completions/mean_length": 79.875, "completions/min_length": 77.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.875, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.9909250736236572, "rewards/meter/std": 0.0028839793521910906, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9494280815124512, "rewards/total_composite/std": 0.11536786705255508, "reward": 0.9494280815124512, "reward_std": 0.11536785960197449, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0175126064568758, "sampling/sampling_logp_difference/max": 2.6809237003326416, "sampling/importance_sampling_ratio/min": 0.0684998482465744, "sampling/importance_sampling_ratio/mean": 1.0048216581344604, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09262991230934858, "clip_ratio/low_mean": 0.004870129749178886, "clip_ratio/low_min": 0.004870129749178886, "clip_ratio/high_mean": 0.01242569915484637, "clip_ratio/high_max": 0.01242569915484637, "clip_ratio/region_mean": 0.017295828904025257, "reward_total_mean": 0.9494280815124512, "reward_meter_mean": 0.9909250736236572, "reward_meter_std": 0.0028839793521910906, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.9494280815124512, "reward_total_composite_std": 0.11536786705255508} {"timestamp_utc": "2026-04-11T23:52:08Z", "mode": "eval", "global_step": 1250, "epoch": 0.05020685223119251, "eval_loss": NaN, "eval_runtime": 56.3598, "eval_samples_per_second": 1.845, "eval_steps_per_second": 0.231, "eval_num_tokens": 2819322.0, "eval_completions/mean_length": 172.58653846153845, "eval_completions/min_length": 65.3076923076923, "eval_completions/max_length": 286.84615384615387, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 172.58653846153845, "eval_completions/min_terminated_length": 65.3076923076923, "eval_completions/max_terminated_length": 286.84615384615387, "eval_rewards/meter/mean": 0.689526264484112, "eval_rewards/meter/std": 0.4130670749224149, "eval_rewards/count_adherence/mean": 0.8135509353417617, "eval_rewards/count_adherence/std": 0.17685549706220627, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/repeat_penalty/mean": 0.8221937005336468, "eval_rewards/repeat_penalty/std": 0.12831746328335542, "eval_rewards/total_composite/mean": 0.47181554253284747, "eval_rewards/total_composite/std": 0.3225762511675174, "eval_reward": 0.47181554253284747, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.012653420428530527, "eval_sampling/sampling_logp_difference/max": 0.8327025633591872, "eval_sampling/importance_sampling_ratio/min": 0.4489339819321266, "eval_sampling/importance_sampling_ratio/mean": 1.002467971581679, "eval_sampling/importance_sampling_ratio/max": 1.3588372377248912, "eval_entropy": 0.10610828204796864, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.47181554253284747, "eval_reward_meter_mean": 0.689526264484112, "eval_reward_meter_std": 0.4130670749224149, "eval_reward_count_adherence_mean": 0.8135509353417617, "eval_reward_count_adherence_std": 0.17685549706220627, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_repeat_penalty_mean": 0.8221937005336468, "eval_reward_repeat_penalty_std": 0.12831746328335542, "eval_reward_total_composite_mean": 0.47181554253284747, "eval_reward_total_composite_std": 0.3225762511675174} {"timestamp_utc": "2026-04-11T23:52:17Z", "mode": "train", "global_step": 1251, "epoch": 0.05024701771297747, "loss": 0.0013, "grad_norm": 6.843512058258057, "learning_rate": 6.212121212121213e-06, "num_tokens": 2821356.0, "completions/mean_length": 99.25, "completions/min_length": 96.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.25, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.9614334106445312, "rewards/meter/std": 0.015211360529065132, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9249999523162842, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.8895110487937927, "rewards/total_composite/std": 0.10235293209552765, "reward": 0.8895110487937927, "reward_std": 0.10235292464494705, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.039652157574892044, "sampling/sampling_logp_difference/max": 3.9673361778259277, "sampling/importance_sampling_ratio/min": 0.018923774361610413, "sampling/importance_sampling_ratio/mean": 0.9989413022994995, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09751791087910533, "clip_ratio/low_mean": 0.007605619612149894, "clip_ratio/low_min": 0.007605619612149894, "clip_ratio/high_mean": 0.01761134958360344, "clip_ratio/high_max": 0.01761134958360344, "clip_ratio/region_mean": 0.025216969195753336, "reward_total_mean": 0.8895110487937927, "reward_meter_mean": 0.9614334106445312, "reward_meter_std": 0.015211360529065132, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9249999523162842, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.8895110487937927, "reward_total_composite_std": 0.10235293209552765} {"timestamp_utc": "2026-04-11T23:52:23Z", "mode": "train", "global_step": 1252, "epoch": 0.05028718319476242, "loss": 0.0046, "grad_norm": 1.9272445440292358, "learning_rate": 6.209090909090909e-06, "num_tokens": 2824023.0, "completions/mean_length": 147.375, "completions/min_length": 143.0, "completions/max_length": 152.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 147.375, "completions/min_terminated_length": 143.0, "completions/max_terminated_length": 152.0, "rewards/meter/mean": 0.9962889552116394, "rewards/meter/std": 0.0008069492760114372, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.7294289469718933, "rewards/total_composite/std": 0.050382453948259354, "reward": 0.7294289469718933, "reward_std": 0.050382472574710846, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014175360091030598, "sampling/sampling_logp_difference/max": 1.445974588394165, "sampling/importance_sampling_ratio/min": 0.23551644384860992, "sampling/importance_sampling_ratio/mean": 1.0037672519683838, "sampling/importance_sampling_ratio/max": 1.5003139972686768, "entropy": 0.09728506673127413, "clip_ratio/low_mean": 0.01109143253415823, "clip_ratio/low_min": 0.01109143253415823, "clip_ratio/high_mean": 0.0025510203558951616, "clip_ratio/high_max": 0.0025510203558951616, "clip_ratio/region_mean": 0.013642452890053391, "reward_total_mean": 0.7294289469718933, "reward_meter_mean": 0.9962889552116394, "reward_meter_std": 0.0008069492760114372, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.7294289469718933, "reward_total_composite_std": 0.050382453948259354} {"timestamp_utc": "2026-04-11T23:52:31Z", "mode": "train", "global_step": 1253, "epoch": 0.050327348676547375, "loss": -0.0071, "grad_norm": 2.051161050796509, "learning_rate": 6.206060606060606e-06, "num_tokens": 2828134.0, "completions/mean_length": 290.875, "completions/min_length": 269.0, "completions/max_length": 313.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 290.875, "completions/min_terminated_length": 269.0, "completions/max_terminated_length": 313.0, "rewards/meter/mean": 0.9930562376976013, "rewards/meter/std": 0.01025536097586155, "rewards/count_adherence/mean": 0.5576923489570618, "rewards/count_adherence/std": 0.03560846298933029, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6976648569107056, "rewards/repeat_penalty/std": 0.03868091478943825, "rewards/total_composite/mean": 0.3865600824356079, "rewards/total_composite/std": 0.0352591797709465, "reward": 0.3865600824356079, "reward_std": 0.0352591797709465, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02420906163752079, "sampling/sampling_logp_difference/max": 2.5028674602508545, "sampling/importance_sampling_ratio/min": 0.11250852793455124, "sampling/importance_sampling_ratio/mean": 0.9969112277030945, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09894186165183783, "clip_ratio/low_mean": 0.005023152916692197, "clip_ratio/low_min": 0.005023152916692197, "clip_ratio/high_mean": 0.008442942867986858, "clip_ratio/high_max": 0.008442942867986858, "clip_ratio/region_mean": 0.013466095784679055, "reward_total_mean": 0.3865600824356079, "reward_meter_mean": 0.9930562376976013, "reward_meter_std": 0.01025536097586155, "reward_count_adherence_mean": 0.5576923489570618, "reward_count_adherence_std": 0.03560846298933029, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6976648569107056, "reward_repeat_penalty_std": 0.03868091478943825, "reward_total_composite_mean": 0.3865600824356079, "reward_total_composite_std": 0.0352591797709465} {"timestamp_utc": "2026-04-11T23:52:36Z", "mode": "train", "global_step": 1254, "epoch": 0.05036751415833233, "loss": -0.0038, "grad_norm": 4.324112892150879, "learning_rate": 6.203030303030304e-06, "num_tokens": 2829964.0, "completions/mean_length": 67.75, "completions/min_length": 67.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.75, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9599640965461731, "rewards/meter/std": 0.03301013261079788, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9599640965461731, "rewards/total_composite/std": 0.03301013261079788, "reward": 0.9599640965461731, "reward_std": 0.03301013633608818, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016246721148490906, "sampling/sampling_logp_difference/max": 2.111745834350586, "sampling/importance_sampling_ratio/min": 0.12102648615837097, "sampling/importance_sampling_ratio/mean": 1.0017342567443848, "sampling/importance_sampling_ratio/max": 1.6136887073516846, "entropy": 0.052088204305619, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/high_mean": 0.00916612590663135, "clip_ratio/high_max": 0.00916612590663135, "clip_ratio/region_mean": 0.011031797504983842, "reward_total_mean": 0.9599640965461731, "reward_meter_mean": 0.9599640965461731, "reward_meter_std": 0.03301013261079788, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9599640965461731, "reward_total_composite_std": 0.03301013261079788} {"timestamp_utc": "2026-04-11T23:52:41Z", "mode": "train", "global_step": 1255, "epoch": 0.05040767964011728, "loss": -0.0, "grad_norm": 4.520086765289307, "learning_rate": 6.200000000000001e-06, "num_tokens": 2832143.0, "completions/mean_length": 95.375, "completions/min_length": 92.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.375, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.6753093600273132, "rewards/meter/std": 0.18275485932826996, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.5924432277679443, "rewards/total_composite/std": 0.18791036307811737, "reward": 0.5924432277679443, "reward_std": 0.18791036307811737, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.036074861884117126, "sampling/sampling_logp_difference/max": 2.096856117248535, "sampling/importance_sampling_ratio/min": 0.12284202128648758, "sampling/importance_sampling_ratio/mean": 1.0073623657226562, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12418541312217712, "clip_ratio/low_mean": 0.015763227711431682, "clip_ratio/low_min": 0.015763227711431682, "clip_ratio/high_mean": 0.015558094717562199, "clip_ratio/high_max": 0.015558094717562199, "clip_ratio/region_mean": 0.03132132242899388, "reward_total_mean": 0.5924432277679443, "reward_meter_mean": 0.6753093600273132, "reward_meter_std": 0.18275485932826996, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.5924432277679443, "reward_total_composite_std": 0.18791036307811737} {"timestamp_utc": "2026-04-11T23:52:46Z", "mode": "train", "global_step": 1256, "epoch": 0.050447845121902236, "loss": 0.016, "grad_norm": 4.7573018074035645, "learning_rate": 6.196969696969698e-06, "num_tokens": 2833894.0, "completions/mean_length": 64.875, "completions/min_length": 63.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.875, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.34324562549591064, "rewards/meter/std": 0.38487473130226135, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.34324562549591064, "rewards/total_composite/std": 0.38487473130226135, "reward": 0.34324562549591064, "reward_std": 0.38487470149993896, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06118984892964363, "sampling/sampling_logp_difference/max": 2.886340618133545, "sampling/importance_sampling_ratio/min": 0.05577996000647545, "sampling/importance_sampling_ratio/mean": 1.0009076595306396, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3142077159136534, "clip_ratio/low_mean": 0.024938606191426516, "clip_ratio/low_min": 0.024938606191426516, "clip_ratio/high_mean": 0.015424644108861685, "clip_ratio/high_max": 0.015424644108861685, "clip_ratio/region_mean": 0.0403632503002882, "reward_total_mean": 0.34324562549591064, "reward_meter_mean": 0.34324562549591064, "reward_meter_std": 0.38487473130226135, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.34324562549591064, "reward_total_composite_std": 0.38487473130226135} {"timestamp_utc": "2026-04-11T23:52:51Z", "mode": "train", "global_step": 1257, "epoch": 0.05048801060368719, "loss": 0.0292, "grad_norm": 11.342290878295898, "learning_rate": 6.1939393939393944e-06, "num_tokens": 2835400.0, "completions/mean_length": 33.25, "completions/min_length": 32.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.25, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.5455847978591919, "rewards/meter/std": 0.46300008893013, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5455847978591919, "rewards/total_composite/std": 0.46300008893013, "reward": 0.5455847978591919, "reward_std": 0.4630000591278076, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05044575408101082, "sampling/sampling_logp_difference/max": 0.9388256072998047, "sampling/importance_sampling_ratio/min": 0.3910868465900421, "sampling/importance_sampling_ratio/mean": 1.0081783533096313, "sampling/importance_sampling_ratio/max": 1.4548543691635132, "entropy": 0.2989681512117386, "clip_ratio/low_mean": 0.03709893091581762, "clip_ratio/low_min": 0.03709893091581762, "clip_ratio/high_mean": 0.011482007801532745, "clip_ratio/high_max": 0.011482007801532745, "clip_ratio/region_mean": 0.048580938717350364, "reward_total_mean": 0.5455847978591919, "reward_meter_mean": 0.5455847978591919, "reward_meter_std": 0.46300008893013, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5455847978591919, "reward_total_composite_std": 0.46300008893013} {"timestamp_utc": "2026-04-11T23:52:58Z", "mode": "train", "global_step": 1258, "epoch": 0.050528176085472144, "loss": 0.003, "grad_norm": 1.4305702447891235, "learning_rate": 6.190909090909092e-06, "num_tokens": 2839026.0, "completions/mean_length": 265.25, "completions/min_length": 259.0, "completions/max_length": 270.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 265.25, "completions/min_terminated_length": 259.0, "completions/max_terminated_length": 270.0, "rewards/meter/mean": 0.9982133507728577, "rewards/meter/std": 0.0003504555206745863, "rewards/count_adherence/mean": 0.6363636255264282, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.701923131942749, "rewards/repeat_penalty/std": 0.027196412906050682, "rewards/total_composite/mean": 0.4458819031715393, "rewards/total_composite/std": 0.017329318448901176, "reward": 0.4458819031715393, "reward_std": 0.017329316586256027, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011669746600091457, "sampling/sampling_logp_difference/max": 1.233236312866211, "sampling/importance_sampling_ratio/min": 0.2913481593132019, "sampling/importance_sampling_ratio/mean": 1.0004147291183472, "sampling/importance_sampling_ratio/max": 1.5016297101974487, "entropy": 0.06276637688279152, "clip_ratio/low_mean": 0.006089399073971435, "clip_ratio/low_min": 0.006089399073971435, "clip_ratio/high_mean": 0.0009433962404727936, "clip_ratio/high_max": 0.0009433962404727936, "clip_ratio/region_mean": 0.007032795314444229, "reward_total_mean": 0.4458819031715393, "reward_meter_mean": 0.9982133507728577, "reward_meter_std": 0.0003504555206745863, "reward_count_adherence_mean": 0.6363636255264282, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.701923131942749, "reward_repeat_penalty_std": 0.027196412906050682, "reward_total_composite_mean": 0.4458819031715393, "reward_total_composite_std": 0.017329318448901176} {"timestamp_utc": "2026-04-11T23:53:03Z", "mode": "train", "global_step": 1259, "epoch": 0.0505683415672571, "loss": 0.013, "grad_norm": 3.326974630355835, "learning_rate": 6.187878787878788e-06, "num_tokens": 2841255.0, "completions/mean_length": 107.625, "completions/min_length": 102.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.625, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.5407921671867371, "rewards/meter/std": 0.3177524209022522, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.925000011920929, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.4992244839668274, "rewards/total_composite/std": 0.31581348180770874, "reward": 0.4992244839668274, "reward_std": 0.31581348180770874, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04232693091034889, "sampling/sampling_logp_difference/max": 1.4409921169281006, "sampling/importance_sampling_ratio/min": 0.24779127538204193, "sampling/importance_sampling_ratio/mean": 1.0075656175613403, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.18995575420558453, "clip_ratio/low_mean": 0.01721830479800701, "clip_ratio/low_min": 0.01721830479800701, "clip_ratio/high_mean": 0.015308146830648184, "clip_ratio/high_max": 0.015308146830648184, "clip_ratio/region_mean": 0.032526451628655195, "reward_total_mean": 0.4992244839668274, "reward_meter_mean": 0.5407921671867371, "reward_meter_std": 0.3177524209022522, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.925000011920929, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.4992244839668274, "reward_total_composite_std": 0.31581348180770874} {"timestamp_utc": "2026-04-11T23:53:08Z", "mode": "train", "global_step": 1260, "epoch": 0.05060850704904205, "loss": 0.0127, "grad_norm": 2.4279935359954834, "learning_rate": 6.184848484848485e-06, "num_tokens": 2842846.0, "completions/mean_length": 40.875, "completions/min_length": 38.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.875, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.9946430921554565, "rewards/meter/std": 0.006536941975355148, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9946430921554565, "rewards/total_composite/std": 0.006536941975355148, "reward": 0.9946430921554565, "reward_std": 0.00653693825006485, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03152947872877121, "sampling/sampling_logp_difference/max": 1.162613868713379, "sampling/importance_sampling_ratio/min": 0.3126678168773651, "sampling/importance_sampling_ratio/mean": 0.9979214072227478, "sampling/importance_sampling_ratio/max": 1.3470659255981445, "entropy": 0.174057693220675, "clip_ratio/low_mean": 0.0029761905316263437, "clip_ratio/low_min": 0.0029761905316263437, "clip_ratio/high_mean": 0.028182906797155738, "clip_ratio/high_max": 0.028182906797155738, "clip_ratio/region_mean": 0.03115909732878208, "reward_total_mean": 0.9946430921554565, "reward_meter_mean": 0.9946430921554565, "reward_meter_std": 0.006536941975355148, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9946430921554565, "reward_total_composite_std": 0.006536941975355148} {"timestamp_utc": "2026-04-11T23:53:13Z", "mode": "train", "global_step": 1261, "epoch": 0.050648672530827006, "loss": 0.0029, "grad_norm": 5.053354263305664, "learning_rate": 6.181818181818182e-06, "num_tokens": 2844913.0, "completions/mean_length": 93.375, "completions/min_length": 91.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.375, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.39358216524124146, "rewards/meter/std": 0.3734220564365387, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.3705918788909912, "rewards/total_composite/std": 0.34063175320625305, "reward": 0.3705918788909912, "reward_std": 0.34063172340393066, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.052313853055238724, "sampling/sampling_logp_difference/max": 1.3460525274276733, "sampling/importance_sampling_ratio/min": 0.26026561856269836, "sampling/importance_sampling_ratio/mean": 1.0092734098434448, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.34710839204490185, "clip_ratio/low_mean": 0.024214978329837322, "clip_ratio/low_min": 0.024214978329837322, "clip_ratio/high_mean": 0.027944298926740885, "clip_ratio/high_max": 0.027944298926740885, "clip_ratio/region_mean": 0.05215927725657821, "reward_total_mean": 0.3705918788909912, "reward_meter_mean": 0.39358216524124146, "reward_meter_std": 0.3734220564365387, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.3705918788909912, "reward_total_composite_std": 0.34063175320625305} {"timestamp_utc": "2026-04-11T23:53:19Z", "mode": "train", "global_step": 1262, "epoch": 0.05068883801261196, "loss": 0.0561, "grad_norm": 2.785076856613159, "learning_rate": 6.17878787878788e-06, "num_tokens": 2847577.0, "completions/mean_length": 158.0, "completions/min_length": 143.0, "completions/max_length": 167.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 158.0, "completions/min_terminated_length": 143.0, "completions/max_terminated_length": 167.0, "rewards/meter/mean": 0.504827618598938, "rewards/meter/std": 0.3096640408039093, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7850378751754761, "rewards/repeat_penalty/std": 0.07399878650903702, "rewards/total_composite/mean": 0.32526856660842896, "rewards/total_composite/std": 0.17344774305820465, "reward": 0.32526856660842896, "reward_std": 0.17344774305820465, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028311705216765404, "sampling/sampling_logp_difference/max": 1.1957664489746094, "sampling/importance_sampling_ratio/min": 0.3024720251560211, "sampling/importance_sampling_ratio/mean": 0.9964918494224548, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0898361736908555, "clip_ratio/low_mean": 0.010547005338594317, "clip_ratio/low_min": 0.010547005338594317, "clip_ratio/high_mean": 0.018799963407218456, "clip_ratio/high_max": 0.018799963407218456, "clip_ratio/region_mean": 0.029346968745812774, "reward_total_mean": 0.32526856660842896, "reward_meter_mean": 0.504827618598938, "reward_meter_std": 0.3096640408039093, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7850378751754761, "reward_repeat_penalty_std": 0.07399878650903702, "reward_total_composite_mean": 0.32526856660842896, "reward_total_composite_std": 0.17344774305820465} {"timestamp_utc": "2026-04-11T23:53:27Z", "mode": "train", "global_step": 1263, "epoch": 0.050729003494396914, "loss": 0.0009, "grad_norm": 1.8219730854034424, "learning_rate": 6.175757575757576e-06, "num_tokens": 2851398.0, "completions/mean_length": 272.625, "completions/min_length": 262.0, "completions/max_length": 285.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 272.625, "completions/min_terminated_length": 262.0, "completions/max_terminated_length": 285.0, "rewards/meter/mean": 0.9851720929145813, "rewards/meter/std": 0.025058675557374954, "rewards/count_adherence/mean": 0.6363636255264282, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7403846383094788, "rewards/repeat_penalty/std": 0.07047118246555328, "rewards/total_composite/mean": 0.4640815854072571, "rewards/total_composite/std": 0.0445028655230999, "reward": 0.4640815854072571, "reward_std": 0.044502876698970795, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023550348356366158, "sampling/sampling_logp_difference/max": 1.8511247634887695, "sampling/importance_sampling_ratio/min": 0.1570604145526886, "sampling/importance_sampling_ratio/mean": 0.9996957182884216, "sampling/importance_sampling_ratio/max": 1.9002686738967896, "entropy": 0.11504366714507341, "clip_ratio/low_mean": 0.007349682127824053, "clip_ratio/low_min": 0.007349682127824053, "clip_ratio/high_mean": 0.006790850544348359, "clip_ratio/high_max": 0.006790850544348359, "clip_ratio/region_mean": 0.014140532672172412, "reward_total_mean": 0.4640815854072571, "reward_meter_mean": 0.9851720929145813, "reward_meter_std": 0.025058675557374954, "reward_count_adherence_mean": 0.6363636255264282, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7403846383094788, "reward_repeat_penalty_std": 0.07047118246555328, "reward_total_composite_mean": 0.4640815854072571, "reward_total_composite_std": 0.0445028655230999} {"timestamp_utc": "2026-04-11T23:53:31Z", "mode": "train", "global_step": 1264, "epoch": 0.05076916897618187, "loss": 0.0151, "grad_norm": 6.026924133300781, "learning_rate": 6.1727272727272735e-06, "num_tokens": 2853090.0, "completions/mean_length": 58.5, "completions/min_length": 56.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.5, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.2589147090911865, "rewards/meter/std": 0.21604736149311066, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.2589147090911865, "rewards/total_composite/std": 0.21604736149311066, "reward": 0.2589147090911865, "reward_std": 0.21604733169078827, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.039983682334423065, "sampling/sampling_logp_difference/max": 0.9148893356323242, "sampling/importance_sampling_ratio/min": 0.4005609452724457, "sampling/importance_sampling_ratio/mean": 1.0070089101791382, "sampling/importance_sampling_ratio/max": 1.932633399963379, "entropy": 0.23071717843413353, "clip_ratio/low_mean": 0.021268213167786598, "clip_ratio/low_min": 0.021268213167786598, "clip_ratio/high_mean": 0.00657894741743803, "clip_ratio/high_max": 0.00657894741743803, "clip_ratio/region_mean": 0.02784716058522463, "reward_total_mean": 0.2589147090911865, "reward_meter_mean": 0.2589147090911865, "reward_meter_std": 0.21604736149311066, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.2589147090911865, "reward_total_composite_std": 0.21604736149311066} {"timestamp_utc": "2026-04-11T23:53:36Z", "mode": "train", "global_step": 1265, "epoch": 0.05080933445796682, "loss": 0.0293, "grad_norm": 16.352861404418945, "learning_rate": 6.16969696969697e-06, "num_tokens": 2854666.0, "completions/mean_length": 36.0, "completions/min_length": 34.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.8565147519111633, "rewards/meter/std": 0.3440394103527069, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8565147519111633, "rewards/total_composite/std": 0.3440394103527069, "reward": 0.8565147519111633, "reward_std": 0.3440393805503845, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.057592298835515976, "sampling/sampling_logp_difference/max": 1.1175026893615723, "sampling/importance_sampling_ratio/min": 0.327095627784729, "sampling/importance_sampling_ratio/mean": 0.9966188073158264, "sampling/importance_sampling_ratio/max": 1.7478100061416626, "entropy": 0.4582943990826607, "clip_ratio/low_mean": 0.0033783784601837397, "clip_ratio/low_min": 0.0033783784601837397, "clip_ratio/high_mean": 0.05259183724410832, "clip_ratio/high_max": 0.05259183724410832, "clip_ratio/region_mean": 0.05597021570429206, "reward_total_mean": 0.8565147519111633, "reward_meter_mean": 0.8565147519111633, "reward_meter_std": 0.3440394103527069, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8565147519111633, "reward_total_composite_std": 0.3440394103527069} {"timestamp_utc": "2026-04-11T23:53:40Z", "mode": "train", "global_step": 1266, "epoch": 0.050849499939751776, "loss": 0.0058, "grad_norm": 3.7630727291107178, "learning_rate": 6.166666666666667e-06, "num_tokens": 2856589.0, "completions/mean_length": 77.375, "completions/min_length": 75.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.375, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9948470592498779, "rewards/meter/std": 0.0029209780041128397, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9948470592498779, "rewards/total_composite/std": 0.0029209780041128397, "reward": 0.9948470592498779, "reward_std": 0.0029209787026047707, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04110397398471832, "sampling/sampling_logp_difference/max": 1.2458505630493164, "sampling/importance_sampling_ratio/min": 0.28769612312316895, "sampling/importance_sampling_ratio/mean": 1.0063776969909668, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2823401764035225, "clip_ratio/low_mean": 0.025986650260165334, "clip_ratio/low_min": 0.025986650260165334, "clip_ratio/high_mean": 0.016067266347818077, "clip_ratio/high_max": 0.016067266347818077, "clip_ratio/region_mean": 0.04205391660798341, "reward_total_mean": 0.9948470592498779, "reward_meter_mean": 0.9948470592498779, "reward_meter_std": 0.0029209780041128397, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9948470592498779, "reward_total_composite_std": 0.0029209780041128397} {"timestamp_utc": "2026-04-11T23:53:45Z", "mode": "train", "global_step": 1267, "epoch": 0.05088966542153673, "loss": 0.0264, "grad_norm": 7.6448774337768555, "learning_rate": 6.163636363636364e-06, "num_tokens": 2858420.0, "completions/mean_length": 63.875, "completions/min_length": 58.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.875, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.6281806230545044, "rewards/meter/std": 0.4045161306858063, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6281806230545044, "rewards/total_composite/std": 0.4045161306858063, "reward": 0.6281806230545044, "reward_std": 0.4045161306858063, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08960650116205215, "sampling/sampling_logp_difference/max": 1.5403270721435547, "sampling/importance_sampling_ratio/min": 0.21431100368499756, "sampling/importance_sampling_ratio/mean": 1.0073941946029663, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5581622421741486, "clip_ratio/low_mean": 0.02490918291732669, "clip_ratio/low_min": 0.02490918291732669, "clip_ratio/high_mean": 0.035607312340289354, "clip_ratio/high_max": 0.035607312340289354, "clip_ratio/region_mean": 0.06051649525761604, "reward_total_mean": 0.6281806230545044, "reward_meter_mean": 0.6281806230545044, "reward_meter_std": 0.4045161306858063, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6281806230545044, "reward_total_composite_std": 0.4045161306858063} {"timestamp_utc": "2026-04-11T23:53:50Z", "mode": "train", "global_step": 1268, "epoch": 0.050929830903321684, "loss": -0.008, "grad_norm": 6.307591438293457, "learning_rate": 6.160606060606062e-06, "num_tokens": 2860190.0, "completions/mean_length": 76.25, "completions/min_length": 73.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.25, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9219943881034851, "rewards/meter/std": 0.2038099318742752, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9219943881034851, "rewards/total_composite/std": 0.2038099318742752, "reward": 0.9219943881034851, "reward_std": 0.2038099467754364, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03656025603413582, "sampling/sampling_logp_difference/max": 1.101435661315918, "sampling/importance_sampling_ratio/min": 0.3323935270309448, "sampling/importance_sampling_ratio/mean": 0.99973064661026, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1600959748029709, "clip_ratio/low_mean": 0.0034246575087308884, "clip_ratio/low_min": 0.0034246575087308884, "clip_ratio/high_mean": 0.028985073906369507, "clip_ratio/high_max": 0.028985073906369507, "clip_ratio/region_mean": 0.032409731415100396, "reward_total_mean": 0.9219943881034851, "reward_meter_mean": 0.9219943881034851, "reward_meter_std": 0.2038099318742752, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9219943881034851, "reward_total_composite_std": 0.2038099318742752} {"timestamp_utc": "2026-04-11T23:53:55Z", "mode": "train", "global_step": 1269, "epoch": 0.05096999638510664, "loss": 0.0401, "grad_norm": 16.08408546447754, "learning_rate": 6.157575757575758e-06, "num_tokens": 2862222.0, "completions/mean_length": 69.0, "completions/min_length": 66.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.8981978893280029, "rewards/meter/std": 0.22402629256248474, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8981978893280029, "rewards/total_composite/std": 0.22402629256248474, "reward": 0.8981978893280029, "reward_std": 0.22402627766132355, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03928679972887039, "sampling/sampling_logp_difference/max": 2.344562530517578, "sampling/importance_sampling_ratio/min": 0.0958891436457634, "sampling/importance_sampling_ratio/mean": 0.9943932890892029, "sampling/importance_sampling_ratio/max": 1.8017491102218628, "entropy": 0.09085107641294599, "clip_ratio/low_mean": 0.0033783784601837397, "clip_ratio/low_min": 0.0033783784601837397, "clip_ratio/high_mean": 0.034388539381325245, "clip_ratio/high_max": 0.034388539381325245, "clip_ratio/region_mean": 0.037766917841508985, "reward_total_mean": 0.8981978893280029, "reward_meter_mean": 0.8981978893280029, "reward_meter_std": 0.22402629256248474, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8981978893280029, "reward_total_composite_std": 0.22402629256248474} {"timestamp_utc": "2026-04-11T23:54:03Z", "mode": "train", "global_step": 1270, "epoch": 0.05101016186689159, "loss": -0.0005, "grad_norm": 1.1973329782485962, "learning_rate": 6.154545454545455e-06, "num_tokens": 2866664.0, "completions/mean_length": 375.25, "completions/min_length": 359.0, "completions/max_length": 388.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 375.25, "completions/min_terminated_length": 359.0, "completions/max_terminated_length": 388.0, "rewards/meter/mean": 0.9982415437698364, "rewards/meter/std": 0.00024763745022937655, "rewards/count_adherence/mean": 0.5263158082962036, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6842105388641357, "rewards/repeat_penalty/std": 0.07443230599164963, "rewards/total_composite/mean": 0.35947954654693604, "rewards/total_composite/std": 0.039129648357629776, "reward": 0.35947954654693604, "reward_std": 0.039129648357629776, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01408437080681324, "sampling/sampling_logp_difference/max": 1.9516000747680664, "sampling/importance_sampling_ratio/min": 0.14204660058021545, "sampling/importance_sampling_ratio/mean": 1.0016210079193115, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07593974284827709, "clip_ratio/low_mean": 0.004985210340237245, "clip_ratio/low_min": 0.004985210340237245, "clip_ratio/high_mean": 0.006335443118587136, "clip_ratio/high_max": 0.006335443118587136, "clip_ratio/region_mean": 0.011320653458824381, "reward_total_mean": 0.35947954654693604, "reward_meter_mean": 0.9982415437698364, "reward_meter_std": 0.00024763745022937655, "reward_count_adherence_mean": 0.5263158082962036, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6842105388641357, "reward_repeat_penalty_std": 0.07443230599164963, "reward_total_composite_mean": 0.35947954654693604, "reward_total_composite_std": 0.039129648357629776} {"timestamp_utc": "2026-04-11T23:54:08Z", "mode": "train", "global_step": 1271, "epoch": 0.051050327348676545, "loss": 0.0134, "grad_norm": 3.943135976791382, "learning_rate": 6.151515151515152e-06, "num_tokens": 2868449.0, "completions/mean_length": 76.125, "completions/min_length": 75.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.125, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9946237802505493, "rewards/meter/std": 0.0009093984263017774, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9946237802505493, "rewards/total_composite/std": 0.0009093984263017774, "reward": 0.9946237802505493, "reward_std": 0.0009094090783037245, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02133559249341488, "sampling/sampling_logp_difference/max": 0.8648512959480286, "sampling/importance_sampling_ratio/min": 0.4211141765117645, "sampling/importance_sampling_ratio/mean": 1.0047322511672974, "sampling/importance_sampling_ratio/max": 1.8516945838928223, "entropy": 0.13653591368347406, "clip_ratio/low_mean": 0.0032056551426649094, "clip_ratio/low_min": 0.0032056551426649094, "clip_ratio/high_mean": 0.00831140368245542, "clip_ratio/high_max": 0.00831140368245542, "clip_ratio/region_mean": 0.01151705882512033, "reward_total_mean": 0.9946237802505493, "reward_meter_mean": 0.9946237802505493, "reward_meter_std": 0.0009093984263017774, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9946237802505493, "reward_total_composite_std": 0.0009093984263017774} {"timestamp_utc": "2026-04-11T23:54:13Z", "mode": "train", "global_step": 1272, "epoch": 0.0510904928304615, "loss": 0.0295, "grad_norm": 5.4271039962768555, "learning_rate": 6.148484848484849e-06, "num_tokens": 2870337.0, "completions/mean_length": 67.0, "completions/min_length": 63.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.7235432863235474, "rewards/meter/std": 0.3634967803955078, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7235432863235474, "rewards/total_composite/std": 0.3634967803955078, "reward": 0.7235432863235474, "reward_std": 0.3634967505931854, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06481360644102097, "sampling/sampling_logp_difference/max": 3.7088615894317627, "sampling/importance_sampling_ratio/min": 0.024505402892827988, "sampling/importance_sampling_ratio/mean": 0.9991426467895508, "sampling/importance_sampling_ratio/max": 1.7601451873779297, "entropy": 0.26362900249660015, "clip_ratio/low_mean": 0.010900080436840653, "clip_ratio/low_min": 0.010900080436840653, "clip_ratio/high_mean": 0.036419710610061884, "clip_ratio/high_max": 0.036419710610061884, "clip_ratio/region_mean": 0.04731979104690254, "reward_total_mean": 0.7235432863235474, "reward_meter_mean": 0.7235432863235474, "reward_meter_std": 0.3634967803955078, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7235432863235474, "reward_total_composite_std": 0.3634967803955078} {"timestamp_utc": "2026-04-11T23:54:19Z", "mode": "train", "global_step": 1273, "epoch": 0.05113065831224645, "loss": 0.0059, "grad_norm": 3.146674394607544, "learning_rate": 6.1454545454545454e-06, "num_tokens": 2872992.0, "completions/mean_length": 152.875, "completions/min_length": 149.0, "completions/max_length": 157.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 152.875, "completions/min_terminated_length": 149.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.9979577660560608, "rewards/meter/std": 0.0006660494254902005, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.855392336845398, "rewards/total_composite/std": 0.0005708994576707482, "reward": 0.855392336845398, "reward_std": 0.0005708927637897432, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01988840475678444, "sampling/sampling_logp_difference/max": 1.3840652704238892, "sampling/importance_sampling_ratio/min": 0.25055789947509766, "sampling/importance_sampling_ratio/mean": 0.9994745850563049, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10434958059340715, "clip_ratio/low_mean": 0.004043337190523744, "clip_ratio/low_min": 0.004043337190523744, "clip_ratio/high_mean": 0.00823544233571738, "clip_ratio/high_max": 0.00823544233571738, "clip_ratio/region_mean": 0.012278779526241124, "reward_total_mean": 0.855392336845398, "reward_meter_mean": 0.9979577660560608, "reward_meter_std": 0.0006660494254902005, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.855392336845398, "reward_total_composite_std": 0.0005708994576707482} {"timestamp_utc": "2026-04-11T23:54:23Z", "mode": "train", "global_step": 1274, "epoch": 0.05117082379403141, "loss": -0.0028, "grad_norm": 9.43035888671875, "learning_rate": 6.142424242424243e-06, "num_tokens": 2874585.0, "completions/mean_length": 42.125, "completions/min_length": 42.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.125, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.9962718486785889, "rewards/meter/std": 0.0009969644015654922, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9962718486785889, "rewards/total_composite/std": 0.0009969644015654922, "reward": 0.9962718486785889, "reward_std": 0.0009969675447791815, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02699565887451172, "sampling/sampling_logp_difference/max": 0.7795038223266602, "sampling/importance_sampling_ratio/min": 0.547152042388916, "sampling/importance_sampling_ratio/mean": 1.0059658288955688, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1509600691497326, "clip_ratio/low_mean": 0.0029761905316263437, "clip_ratio/low_min": 0.0029761905316263437, "clip_ratio/high_mean": 0.008859357796609402, "clip_ratio/high_max": 0.008859357796609402, "clip_ratio/region_mean": 0.011835548328235745, "reward_total_mean": 0.9962718486785889, "reward_meter_mean": 0.9962718486785889, "reward_meter_std": 0.0009969644015654922, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9962718486785889, "reward_total_composite_std": 0.0009969644015654922} {"timestamp_utc": "2026-04-11T23:54:29Z", "mode": "train", "global_step": 1275, "epoch": 0.05121098927581636, "loss": -0.0071, "grad_norm": 3.5774126052856445, "learning_rate": 6.139393939393939e-06, "num_tokens": 2877026.0, "completions/mean_length": 128.125, "completions/min_length": 121.0, "completions/max_length": 137.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.125, "completions/min_terminated_length": 121.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.9509094953536987, "rewards/meter/std": 0.036813125014305115, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.8314633369445801, "rewards/total_composite/std": 0.04581277072429657, "reward": 0.8314633369445801, "reward_std": 0.04581277072429657, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.035852231085300446, "sampling/sampling_logp_difference/max": 1.1192617416381836, "sampling/importance_sampling_ratio/min": 0.32652077078819275, "sampling/importance_sampling_ratio/mean": 1.0062079429626465, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.20628871954977512, "clip_ratio/low_mean": 0.007767904899083078, "clip_ratio/low_min": 0.007767904899083078, "clip_ratio/high_mean": 0.012963371758814901, "clip_ratio/high_max": 0.012963371758814901, "clip_ratio/region_mean": 0.02073127665789798, "reward_total_mean": 0.8314633369445801, "reward_meter_mean": 0.9509094953536987, "reward_meter_std": 0.036813125014305115, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.8314633369445801, "reward_total_composite_std": 0.04581277072429657} {"timestamp_utc": "2026-04-11T23:54:34Z", "mode": "train", "global_step": 1276, "epoch": 0.051251154757601315, "loss": 0.0392, "grad_norm": 7.507855415344238, "learning_rate": 6.136363636363637e-06, "num_tokens": 2879100.0, "completions/mean_length": 89.25, "completions/min_length": 84.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.25, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.414013534784317, "rewards/meter/std": 0.36701732873916626, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.3967258036136627, "rewards/total_composite/std": 0.3643990457057953, "reward": 0.3967258036136627, "reward_std": 0.3643990457057953, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09024675190448761, "sampling/sampling_logp_difference/max": 1.623225212097168, "sampling/importance_sampling_ratio/min": 0.19726145267486572, "sampling/importance_sampling_ratio/mean": 1.0259032249450684, "sampling/importance_sampling_ratio/max": 1.9257968664169312, "entropy": 0.8716921396553516, "clip_ratio/low_mean": 0.0324383438564837, "clip_ratio/low_min": 0.0324383438564837, "clip_ratio/high_mean": 0.028633129317313433, "clip_ratio/high_max": 0.028633129317313433, "clip_ratio/region_mean": 0.06107147317379713, "reward_total_mean": 0.3967258036136627, "reward_meter_mean": 0.414013534784317, "reward_meter_std": 0.36701732873916626, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.3967258036136627, "reward_total_composite_std": 0.3643990457057953} {"timestamp_utc": "2026-04-11T23:54:40Z", "mode": "train", "global_step": 1277, "epoch": 0.05129132023938627, "loss": 0.0264, "grad_norm": 5.259714126586914, "learning_rate": 6.133333333333334e-06, "num_tokens": 2881970.0, "completions/mean_length": 177.75, "completions/min_length": 170.0, "completions/max_length": 184.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 177.75, "completions/min_terminated_length": 170.0, "completions/max_terminated_length": 184.0, "rewards/meter/mean": 0.8953356742858887, "rewards/meter/std": 0.2748320698738098, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8611111640930176, "rewards/repeat_penalty/std": 0.05143444612622261, "rewards/total_composite/mean": 0.6403263807296753, "rewards/total_composite/std": 0.19905686378479004, "reward": 0.6403263807296753, "reward_std": 0.19905687868595123, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04297341778874397, "sampling/sampling_logp_difference/max": 1.2142763137817383, "sampling/importance_sampling_ratio/min": 0.2969248294830322, "sampling/importance_sampling_ratio/mean": 1.0098189115524292, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3461093604564667, "clip_ratio/low_mean": 0.012976306956261396, "clip_ratio/low_min": 0.012976306956261396, "clip_ratio/high_mean": 0.02701578661799431, "clip_ratio/high_max": 0.02701578661799431, "clip_ratio/region_mean": 0.039992093574255705, "reward_total_mean": 0.6403263807296753, "reward_meter_mean": 0.8953356742858887, "reward_meter_std": 0.2748320698738098, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8611111640930176, "reward_repeat_penalty_std": 0.05143444612622261, "reward_total_composite_mean": 0.6403263807296753, "reward_total_composite_std": 0.19905686378479004} {"timestamp_utc": "2026-04-11T23:54:44Z", "mode": "train", "global_step": 1278, "epoch": 0.05133148572117122, "loss": 0.0461, "grad_norm": 10.583206176757812, "learning_rate": 6.130303030303031e-06, "num_tokens": 2883696.0, "completions/mean_length": 48.75, "completions/min_length": 42.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.75, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.8016507625579834, "rewards/meter/std": 0.21309679746627808, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8016507625579834, "rewards/total_composite/std": 0.21309679746627808, "reward": 0.8016507625579834, "reward_std": 0.21309679746627808, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07192375510931015, "sampling/sampling_logp_difference/max": 1.8451597690582275, "sampling/importance_sampling_ratio/min": 0.15800006687641144, "sampling/importance_sampling_ratio/mean": 0.992801308631897, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2942120973020792, "clip_ratio/low_mean": 0.012557290028780699, "clip_ratio/low_min": 0.012557290028780699, "clip_ratio/high_mean": 0.054509096313267946, "clip_ratio/high_max": 0.054509096313267946, "clip_ratio/region_mean": 0.06706638634204865, "reward_total_mean": 0.8016507625579834, "reward_meter_mean": 0.8016507625579834, "reward_meter_std": 0.21309679746627808, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8016507625579834, "reward_total_composite_std": 0.21309679746627808} {"timestamp_utc": "2026-04-11T23:54:49Z", "mode": "train", "global_step": 1279, "epoch": 0.05137165120295618, "loss": -0.028, "grad_norm": 4.000077247619629, "learning_rate": 6.127272727272727e-06, "num_tokens": 2885855.0, "completions/mean_length": 86.875, "completions/min_length": 79.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.875, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.7521673440933228, "rewards/meter/std": 0.3259941041469574, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.1511857807636261, "rewards/total_composite/mean": 0.625595211982727, "rewards/total_composite/std": 0.32554712891578674, "reward": 0.625595211982727, "reward_std": 0.32554712891578674, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03764592856168747, "sampling/sampling_logp_difference/max": 1.77433443069458, "sampling/importance_sampling_ratio/min": 0.45753392577171326, "sampling/importance_sampling_ratio/mean": 1.005535364151001, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.20182008855044842, "clip_ratio/low_mean": 0.010369318537414074, "clip_ratio/low_min": 0.010369318537414074, "clip_ratio/high_mean": 0.0231478811474517, "clip_ratio/high_max": 0.0231478811474517, "clip_ratio/region_mean": 0.03351719968486577, "reward_total_mean": 0.625595211982727, "reward_meter_mean": 0.7521673440933228, "reward_meter_std": 0.3259941041469574, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.1511857807636261, "reward_total_composite_mean": 0.625595211982727, "reward_total_composite_std": 0.32554712891578674} {"timestamp_utc": "2026-04-11T23:54:54Z", "mode": "train", "global_step": 1280, "epoch": 0.05141181668474113, "loss": 0.0056, "grad_norm": 5.375721454620361, "learning_rate": 6.1242424242424245e-06, "num_tokens": 2888100.0, "completions/mean_length": 110.625, "completions/min_length": 107.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 110.625, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.32401683926582336, "rewards/meter/std": 0.4113844931125641, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.3231015205383301, "rewards/total_composite/std": 0.41212281584739685, "reward": 0.3231015205383301, "reward_std": 0.41212278604507446, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07019690424203873, "sampling/sampling_logp_difference/max": 4.393073081970215, "sampling/importance_sampling_ratio/min": 0.012362679466605186, "sampling/importance_sampling_ratio/mean": 1.0015076398849487, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3459767811000347, "clip_ratio/low_mean": 0.035150347743183374, "clip_ratio/low_min": 0.035150347743183374, "clip_ratio/high_mean": 0.020130750257521868, "clip_ratio/high_max": 0.020130750257521868, "clip_ratio/region_mean": 0.05528109800070524, "reward_total_mean": 0.3231015205383301, "reward_meter_mean": 0.32401683926582336, "reward_meter_std": 0.4113844931125641, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.3231015205383301, "reward_total_composite_std": 0.41212281584739685} {"timestamp_utc": "2026-04-11T23:54:59Z", "mode": "train", "global_step": 1281, "epoch": 0.051451982166526085, "loss": 0.0157, "grad_norm": 7.6968994140625, "learning_rate": 6.121212121212121e-06, "num_tokens": 2889991.0, "completions/mean_length": 70.375, "completions/min_length": 69.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.375, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9908488988876343, "rewards/meter/std": 0.0105263227596879, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9908488988876343, "rewards/total_composite/std": 0.0105263227596879, "reward": 0.9908488988876343, "reward_std": 0.010526325553655624, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.044087208807468414, "sampling/sampling_logp_difference/max": 1.6382884979248047, "sampling/importance_sampling_ratio/min": 0.19431231915950775, "sampling/importance_sampling_ratio/mean": 0.9948356747627258, "sampling/importance_sampling_ratio/max": 1.7191059589385986, "entropy": 0.2260036300867796, "clip_ratio/low_mean": 0.005307539715431631, "clip_ratio/low_min": 0.005307539715431631, "clip_ratio/high_mean": 0.03385740250814706, "clip_ratio/high_max": 0.03385740250814706, "clip_ratio/region_mean": 0.03916494222357869, "reward_total_mean": 0.9908488988876343, "reward_meter_mean": 0.9908488988876343, "reward_meter_std": 0.0105263227596879, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9908488988876343, "reward_total_composite_std": 0.0105263227596879} {"timestamp_utc": "2026-04-11T23:55:07Z", "mode": "train", "global_step": 1282, "epoch": 0.05149214764831104, "loss": 0.0068, "grad_norm": 2.044459581375122, "learning_rate": 6.118181818181819e-06, "num_tokens": 2894526.0, "completions/mean_length": 303.875, "completions/min_length": 290.0, "completions/max_length": 311.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 303.875, "completions/min_terminated_length": 290.0, "completions/max_terminated_length": 311.0, "rewards/meter/mean": 0.9781317114830017, "rewards/meter/std": 0.027103744447231293, "rewards/count_adherence/mean": 0.5714285969734192, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7749999761581421, "rewards/repeat_penalty/std": 0.11233452707529068, "rewards/total_composite/mean": 0.43306678533554077, "rewards/total_composite/std": 0.0634385421872139, "reward": 0.43306678533554077, "reward_std": 0.0634385421872139, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03700267896056175, "sampling/sampling_logp_difference/max": 3.257213592529297, "sampling/importance_sampling_ratio/min": 0.0384955108165741, "sampling/importance_sampling_ratio/mean": 1.0035651922225952, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.20566194783896208, "clip_ratio/low_mean": 0.010263397532980889, "clip_ratio/low_min": 0.010263397532980889, "clip_ratio/high_mean": 0.013975600013509393, "clip_ratio/high_max": 0.013975600013509393, "clip_ratio/region_mean": 0.024238997546490282, "reward_total_mean": 0.43306678533554077, "reward_meter_mean": 0.9781317114830017, "reward_meter_std": 0.027103744447231293, "reward_count_adherence_mean": 0.5714285969734192, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7749999761581421, "reward_repeat_penalty_std": 0.11233452707529068, "reward_total_composite_mean": 0.43306678533554077, "reward_total_composite_std": 0.0634385421872139} {"timestamp_utc": "2026-04-11T23:55:12Z", "mode": "train", "global_step": 1283, "epoch": 0.05153231313009599, "loss": 0.0097, "grad_norm": 5.159550666809082, "learning_rate": 6.115151515151516e-06, "num_tokens": 2896475.0, "completions/mean_length": 76.625, "completions/min_length": 72.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.625, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9953988790512085, "rewards/meter/std": 0.0025027033407241106, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9953988790512085, "rewards/total_composite/std": 0.0025027033407241106, "reward": 0.9953988790512085, "reward_std": 0.0025027082301676273, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.053737834095954895, "sampling/sampling_logp_difference/max": 1.5215466022491455, "sampling/importance_sampling_ratio/min": 0.21837389469146729, "sampling/importance_sampling_ratio/mean": 1.0098578929901123, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43742145597934723, "clip_ratio/low_mean": 0.009698634734377265, "clip_ratio/low_min": 0.009698634734377265, "clip_ratio/high_mean": 0.030571716954000294, "clip_ratio/high_max": 0.030571716954000294, "clip_ratio/region_mean": 0.04027035168837756, "reward_total_mean": 0.9953988790512085, "reward_meter_mean": 0.9953988790512085, "reward_meter_std": 0.0025027033407241106, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9953988790512085, "reward_total_composite_std": 0.0025027033407241106} {"timestamp_utc": "2026-04-11T23:55:16Z", "mode": "train", "global_step": 1284, "epoch": 0.05157247861188095, "loss": -0.0204, "grad_norm": 8.960368156433105, "learning_rate": 6.112121212121213e-06, "num_tokens": 2897859.0, "completions/mean_length": 36.0, "completions/min_length": 34.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.7910915613174438, "rewards/meter/std": 0.36750105023384094, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7910915613174438, "rewards/total_composite/std": 0.36750105023384094, "reward": 0.7910915613174438, "reward_std": 0.36750105023384094, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06175791099667549, "sampling/sampling_logp_difference/max": 1.3975987434387207, "sampling/importance_sampling_ratio/min": 0.24718981981277466, "sampling/importance_sampling_ratio/mean": 0.996502161026001, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31284985691308975, "clip_ratio/low_mean": 0.03277311008423567, "clip_ratio/low_min": 0.03277311008423567, "clip_ratio/high_mean": 0.020760516403242946, "clip_ratio/high_max": 0.020760516403242946, "clip_ratio/region_mean": 0.053533626487478614, "reward_total_mean": 0.7910915613174438, "reward_meter_mean": 0.7910915613174438, "reward_meter_std": 0.36750105023384094, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7910915613174438, "reward_total_composite_std": 0.36750105023384094} {"timestamp_utc": "2026-04-11T23:55:22Z", "mode": "train", "global_step": 1285, "epoch": 0.0516126440936659, "loss": -0.0018, "grad_norm": 4.468361854553223, "learning_rate": 6.10909090909091e-06, "num_tokens": 2900285.0, "completions/mean_length": 133.25, "completions/min_length": 125.0, "completions/max_length": 146.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 133.25, "completions/min_terminated_length": 125.0, "completions/max_terminated_length": 146.0, "rewards/meter/mean": 0.7679275274276733, "rewards/meter/std": 0.3219233751296997, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9285714626312256, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.7135751247406006, "rewards/total_composite/std": 0.3173206150531769, "reward": 0.7135751247406006, "reward_std": 0.31732064485549927, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04117295891046524, "sampling/sampling_logp_difference/max": 1.2831401824951172, "sampling/importance_sampling_ratio/min": 0.27716556191444397, "sampling/importance_sampling_ratio/mean": 1.006422996520996, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.272802772000432, "clip_ratio/low_mean": 0.013498181011527777, "clip_ratio/low_min": 0.013498181011527777, "clip_ratio/high_mean": 0.01694088790100068, "clip_ratio/high_max": 0.01694088790100068, "clip_ratio/region_mean": 0.030439068912528455, "reward_total_mean": 0.7135751247406006, "reward_meter_mean": 0.7679275274276733, "reward_meter_std": 0.3219233751296997, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9285714626312256, "reward_repeat_penalty_std": 0.07636035233736038, "reward_total_composite_mean": 0.7135751247406006, "reward_total_composite_std": 0.3173206150531769} {"timestamp_utc": "2026-04-11T23:55:29Z", "mode": "train", "global_step": 1286, "epoch": 0.051652809575450855, "loss": 0.0163, "grad_norm": 2.972513437271118, "learning_rate": 6.106060606060606e-06, "num_tokens": 2903604.0, "completions/mean_length": 219.875, "completions/min_length": 204.0, "completions/max_length": 233.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 219.875, "completions/min_terminated_length": 204.0, "completions/max_terminated_length": 233.0, "rewards/meter/mean": 0.8396387100219727, "rewards/meter/std": 0.2860766649246216, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8863636255264282, "rewards/repeat_penalty/std": 0.06428244709968567, "rewards/total_composite/mean": 0.6425139904022217, "rewards/total_composite/std": 0.22964133322238922, "reward": 0.6425139904022217, "reward_std": 0.22964133322238922, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0455651730298996, "sampling/sampling_logp_difference/max": 1.4966444969177246, "sampling/importance_sampling_ratio/min": 0.26713767647743225, "sampling/importance_sampling_ratio/mean": 1.0046546459197998, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3109611961990595, "clip_ratio/low_mean": 0.007508100010454655, "clip_ratio/low_min": 0.007508100010454655, "clip_ratio/high_mean": 0.027195949805900455, "clip_ratio/high_max": 0.027195949805900455, "clip_ratio/region_mean": 0.03470404981635511, "reward_total_mean": 0.6425139904022217, "reward_meter_mean": 0.8396387100219727, "reward_meter_std": 0.2860766649246216, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8863636255264282, "reward_repeat_penalty_std": 0.06428244709968567, "reward_total_composite_mean": 0.6425139904022217, "reward_total_composite_std": 0.22964133322238922} {"timestamp_utc": "2026-04-11T23:55:35Z", "mode": "train", "global_step": 1287, "epoch": 0.05169297505723581, "loss": -0.0036, "grad_norm": 3.1675779819488525, "learning_rate": 6.103030303030304e-06, "num_tokens": 2906858.0, "completions/mean_length": 189.75, "completions/min_length": 182.0, "completions/max_length": 199.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 189.75, "completions/min_terminated_length": 182.0, "completions/max_terminated_length": 199.0, "rewards/meter/mean": 0.9928793907165527, "rewards/meter/std": 0.01269504800438881, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8611111640930176, "rewards/repeat_penalty/std": 0.09848947077989578, "rewards/total_composite/mean": 0.7119789123535156, "rewards/total_composite/std": 0.07652789354324341, "reward": 0.7119789123535156, "reward_std": 0.0765279084444046, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0419112965464592, "sampling/sampling_logp_difference/max": 3.135072708129883, "sampling/importance_sampling_ratio/min": 0.04349659010767937, "sampling/importance_sampling_ratio/mean": 1.0034816265106201, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.22143690288066864, "clip_ratio/low_mean": 0.007304175465833396, "clip_ratio/low_min": 0.007304175465833396, "clip_ratio/high_mean": 0.020334920845925808, "clip_ratio/high_max": 0.020334920845925808, "clip_ratio/region_mean": 0.027639096311759204, "reward_total_mean": 0.7119789123535156, "reward_meter_mean": 0.9928793907165527, "reward_meter_std": 0.01269504800438881, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8611111640930176, "reward_repeat_penalty_std": 0.09848947077989578, "reward_total_composite_mean": 0.7119789123535156, "reward_total_composite_std": 0.07652789354324341} {"timestamp_utc": "2026-04-11T23:55:42Z", "mode": "train", "global_step": 1288, "epoch": 0.05173314053902076, "loss": 0.0056, "grad_norm": 1.9611791372299194, "learning_rate": 6.1e-06, "num_tokens": 2910730.0, "completions/mean_length": 272.0, "completions/min_length": 269.0, "completions/max_length": 275.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 272.0, "completions/min_terminated_length": 269.0, "completions/max_terminated_length": 275.0, "rewards/meter/mean": 0.9904348254203796, "rewards/meter/std": 0.015252761542797089, "rewards/count_adherence/mean": 0.7777777910232544, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8365384340286255, "rewards/repeat_penalty/std": 0.08661473542451859, "rewards/total_composite/mean": 0.6442359685897827, "rewards/total_composite/std": 0.06582249701023102, "reward": 0.6442359685897827, "reward_std": 0.06582249701023102, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025431334972381592, "sampling/sampling_logp_difference/max": 2.2604598999023438, "sampling/importance_sampling_ratio/min": 0.10430250316858292, "sampling/importance_sampling_ratio/mean": 1.0058023929595947, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.139354033395648, "clip_ratio/low_mean": 0.008273915416793898, "clip_ratio/low_min": 0.008273915416793898, "clip_ratio/high_mean": 0.00875260157044977, "clip_ratio/high_max": 0.00875260157044977, "clip_ratio/region_mean": 0.017026516987243667, "reward_total_mean": 0.6442359685897827, "reward_meter_mean": 0.9904348254203796, "reward_meter_std": 0.015252761542797089, "reward_count_adherence_mean": 0.7777777910232544, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8365384340286255, "reward_repeat_penalty_std": 0.08661473542451859, "reward_total_composite_mean": 0.6442359685897827, "reward_total_composite_std": 0.06582249701023102} {"timestamp_utc": "2026-04-11T23:55:47Z", "mode": "train", "global_step": 1289, "epoch": 0.051773306020805716, "loss": 0.0464, "grad_norm": 5.8379316329956055, "learning_rate": 6.096969696969698e-06, "num_tokens": 2912411.0, "completions/mean_length": 68.125, "completions/min_length": 62.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.125, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.8378640413284302, "rewards/meter/std": 0.22375915944576263, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8378640413284302, "rewards/total_composite/std": 0.22375915944576263, "reward": 0.8378640413284302, "reward_std": 0.22375912964344025, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.051838938146829605, "sampling/sampling_logp_difference/max": 1.0195064544677734, "sampling/importance_sampling_ratio/min": 0.360772967338562, "sampling/importance_sampling_ratio/mean": 1.0112100839614868, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.32122630439698696, "clip_ratio/low_mean": 0.01717152213677764, "clip_ratio/low_min": 0.01717152213677764, "clip_ratio/high_mean": 0.0282916008727625, "clip_ratio/high_max": 0.0282916008727625, "clip_ratio/region_mean": 0.04546312300954014, "reward_total_mean": 0.8378640413284302, "reward_meter_mean": 0.8378640413284302, "reward_meter_std": 0.22375915944576263, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8378640413284302, "reward_total_composite_std": 0.22375915944576263} {"timestamp_utc": "2026-04-11T23:55:54Z", "mode": "train", "global_step": 1290, "epoch": 0.05181347150259067, "loss": 0.0012, "grad_norm": 2.28285813331604, "learning_rate": 6.0939393939393946e-06, "num_tokens": 2915945.0, "completions/mean_length": 251.75, "completions/min_length": 245.0, "completions/max_length": 264.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 251.75, "completions/min_terminated_length": 245.0, "completions/max_terminated_length": 264.0, "rewards/meter/mean": 0.017678312957286835, "rewards/meter/std": 0.021944953128695488, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8269230723381042, "rewards/repeat_penalty/std": 0.09859537333250046, "rewards/total_composite/mean": 0.012334112077951431, "rewards/total_composite/std": 0.01488783210515976, "reward": 0.012334112077951431, "reward_std": 0.01488783210515976, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.041879359632730484, "sampling/sampling_logp_difference/max": 1.4119462966918945, "sampling/importance_sampling_ratio/min": 0.2436685711145401, "sampling/importance_sampling_ratio/mean": 1.0024718046188354, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.28565115854144096, "clip_ratio/low_mean": 0.01538592274300754, "clip_ratio/low_min": 0.01538592274300754, "clip_ratio/high_mean": 0.01394439465366304, "clip_ratio/high_max": 0.01394439465366304, "clip_ratio/region_mean": 0.02933031739667058, "reward_total_mean": 0.012334112077951431, "reward_meter_mean": 0.017678312957286835, "reward_meter_std": 0.021944953128695488, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8269230723381042, "reward_repeat_penalty_std": 0.09859537333250046, "reward_total_composite_mean": 0.012334112077951431, "reward_total_composite_std": 0.01488783210515976} {"timestamp_utc": "2026-04-11T23:55:59Z", "mode": "train", "global_step": 1291, "epoch": 0.051853636984375624, "loss": -0.0041, "grad_norm": 3.836810350418091, "learning_rate": 6.090909090909092e-06, "num_tokens": 2918069.0, "completions/mean_length": 102.5, "completions/min_length": 100.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.5, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.9957513809204102, "rewards/meter/std": 0.0030371833126991987, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9708477258682251, "rewards/total_composite/std": 0.07034352421760559, "reward": 0.9708477258682251, "reward_std": 0.07034352421760559, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03093818947672844, "sampling/sampling_logp_difference/max": 0.7555546760559082, "sampling/importance_sampling_ratio/min": 0.4889427423477173, "sampling/importance_sampling_ratio/mean": 1.0092651844024658, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.21113021485507488, "clip_ratio/low_mean": 0.0062500000931322575, "clip_ratio/low_min": 0.0062500000931322575, "clip_ratio/high_mean": 0.024315623799338937, "clip_ratio/high_max": 0.024315623799338937, "clip_ratio/region_mean": 0.030565623892471194, "reward_total_mean": 0.9708477258682251, "reward_meter_mean": 0.9957513809204102, "reward_meter_std": 0.0030371833126991987, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9708477258682251, "reward_total_composite_std": 0.07034352421760559} {"timestamp_utc": "2026-04-11T23:56:04Z", "mode": "train", "global_step": 1292, "epoch": 0.05189380246616058, "loss": 0.0073, "grad_norm": 4.021605968475342, "learning_rate": 6.087878787878788e-06, "num_tokens": 2919873.0, "completions/mean_length": 73.5, "completions/min_length": 71.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.5, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9728573560714722, "rewards/meter/std": 0.0436580516397953, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9728573560714722, "rewards/total_composite/std": 0.0436580516397953, "reward": 0.9728573560714722, "reward_std": 0.0436580590903759, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06107901781797409, "sampling/sampling_logp_difference/max": 1.4575915336608887, "sampling/importance_sampling_ratio/min": 0.23279628157615662, "sampling/importance_sampling_ratio/mean": 1.0169715881347656, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4877397455275059, "clip_ratio/low_mean": 0.006756756920367479, "clip_ratio/low_min": 0.006756756920367479, "clip_ratio/high_mean": 0.03230128774885088, "clip_ratio/high_max": 0.03230128774885088, "clip_ratio/region_mean": 0.03905804466921836, "reward_total_mean": 0.9728573560714722, "reward_meter_mean": 0.9728573560714722, "reward_meter_std": 0.0436580516397953, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9728573560714722, "reward_total_composite_std": 0.0436580516397953} {"timestamp_utc": "2026-04-11T23:56:09Z", "mode": "train", "global_step": 1293, "epoch": 0.05193396794794554, "loss": 0.0401, "grad_norm": 6.868655681610107, "learning_rate": 6.0848484848484855e-06, "num_tokens": 2921458.0, "completions/mean_length": 37.125, "completions/min_length": 36.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.125, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.9967739582061768, "rewards/meter/std": 0.002069387584924698, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9967739582061768, "rewards/total_composite/std": 0.002069387584924698, "reward": 0.9967739582061768, "reward_std": 0.00206937943585217, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04344024881720543, "sampling/sampling_logp_difference/max": 2.1784615516662598, "sampling/importance_sampling_ratio/min": 0.11321556568145752, "sampling/importance_sampling_ratio/mean": 0.9899824857711792, "sampling/importance_sampling_ratio/max": 1.8224308490753174, "entropy": 0.13870495092123747, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.027407341869547963, "clip_ratio/high_max": 0.027407341869547963, "clip_ratio/region_mean": 0.027407341869547963, "reward_total_mean": 0.9967739582061768, "reward_meter_mean": 0.9967739582061768, "reward_meter_std": 0.002069387584924698, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9967739582061768, "reward_total_composite_std": 0.002069387584924698} {"timestamp_utc": "2026-04-11T23:56:14Z", "mode": "train", "global_step": 1294, "epoch": 0.05197413342973049, "loss": 0.0191, "grad_norm": 4.397799491882324, "learning_rate": 6.081818181818182e-06, "num_tokens": 2923230.0, "completions/mean_length": 67.5, "completions/min_length": 64.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.5, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.938705325126648, "rewards/meter/std": 0.06307699531316757, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.938705325126648, "rewards/total_composite/std": 0.06307699531316757, "reward": 0.938705325126648, "reward_std": 0.06307698041200638, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05176489055156708, "sampling/sampling_logp_difference/max": 2.194624423980713, "sampling/importance_sampling_ratio/min": 0.11140039563179016, "sampling/importance_sampling_ratio/mean": 1.0048556327819824, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.30966538563370705, "clip_ratio/low_mean": 0.005599473137408495, "clip_ratio/low_min": 0.005599473137408495, "clip_ratio/high_mean": 0.031401457847096026, "clip_ratio/high_max": 0.031401457847096026, "clip_ratio/region_mean": 0.03700093098450452, "reward_total_mean": 0.938705325126648, "reward_meter_mean": 0.938705325126648, "reward_meter_std": 0.06307699531316757, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.938705325126648, "reward_total_composite_std": 0.06307699531316757} {"timestamp_utc": "2026-04-11T23:56:19Z", "mode": "train", "global_step": 1295, "epoch": 0.05201429891151545, "loss": 0.0019, "grad_norm": 7.243402481079102, "learning_rate": 6.07878787878788e-06, "num_tokens": 2924989.0, "completions/mean_length": 66.875, "completions/min_length": 62.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.875, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9284621477127075, "rewards/meter/std": 0.13208235800266266, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9284621477127075, "rewards/total_composite/std": 0.13208235800266266, "reward": 0.9284621477127075, "reward_std": 0.13208235800266266, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05147900432348251, "sampling/sampling_logp_difference/max": 1.085993766784668, "sampling/importance_sampling_ratio/min": 0.33756616711616516, "sampling/importance_sampling_ratio/mean": 1.0104353427886963, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3127963822335005, "clip_ratio/low_mean": 0.011194029822945595, "clip_ratio/low_min": 0.011194029822945595, "clip_ratio/high_mean": 0.032499466673471034, "clip_ratio/high_max": 0.032499466673471034, "clip_ratio/region_mean": 0.04369349649641663, "reward_total_mean": 0.9284621477127075, "reward_meter_mean": 0.9284621477127075, "reward_meter_std": 0.13208235800266266, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9284621477127075, "reward_total_composite_std": 0.13208235800266266} {"timestamp_utc": "2026-04-11T23:56:24Z", "mode": "train", "global_step": 1296, "epoch": 0.0520544643933004, "loss": 0.0071, "grad_norm": 5.327840328216553, "learning_rate": 6.0757575757575755e-06, "num_tokens": 2927167.0, "completions/mean_length": 108.25, "completions/min_length": 104.0, "completions/max_length": 115.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.25, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 115.0, "rewards/meter/mean": 0.8149079084396362, "rewards/meter/std": 0.24847427010536194, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8149079084396362, "rewards/total_composite/std": 0.24847427010536194, "reward": 0.8149079084396362, "reward_std": 0.24847425520420074, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06685814261436462, "sampling/sampling_logp_difference/max": 1.5687246322631836, "sampling/importance_sampling_ratio/min": 0.20831067860126495, "sampling/importance_sampling_ratio/mean": 1.0199620723724365, "sampling/importance_sampling_ratio/max": 1.9841198921203613, "entropy": 0.6056771166622639, "clip_ratio/low_mean": 0.01175213698297739, "clip_ratio/low_min": 0.01175213698297739, "clip_ratio/high_mean": 0.04434093181043863, "clip_ratio/high_max": 0.04434093181043863, "clip_ratio/region_mean": 0.05609306879341602, "reward_total_mean": 0.8149079084396362, "reward_meter_mean": 0.8149079084396362, "reward_meter_std": 0.24847427010536194, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8149079084396362, "reward_total_composite_std": 0.24847427010536194} {"timestamp_utc": "2026-04-11T23:56:29Z", "mode": "train", "global_step": 1297, "epoch": 0.052094629875085355, "loss": 0.019, "grad_norm": 4.677238941192627, "learning_rate": 6.072727272727274e-06, "num_tokens": 2929000.0, "completions/mean_length": 73.125, "completions/min_length": 71.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.125, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9951352477073669, "rewards/meter/std": 0.0018285271944478154, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9951352477073669, "rewards/total_composite/std": 0.0018285271944478154, "reward": 0.9951352477073669, "reward_std": 0.0018285303376615047, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0533045269548893, "sampling/sampling_logp_difference/max": 2.0506200790405273, "sampling/importance_sampling_ratio/min": 0.1286551058292389, "sampling/importance_sampling_ratio/mean": 1.010929822921753, "sampling/importance_sampling_ratio/max": 1.4575989246368408, "entropy": 0.3857220932841301, "clip_ratio/low_mean": 0.015135752153582871, "clip_ratio/low_min": 0.015135752153582871, "clip_ratio/high_mean": 0.02069655992090702, "clip_ratio/high_max": 0.02069655992090702, "clip_ratio/region_mean": 0.03583231207448989, "reward_total_mean": 0.9951352477073669, "reward_meter_mean": 0.9951352477073669, "reward_meter_std": 0.0018285271944478154, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9951352477073669, "reward_total_composite_std": 0.0018285271944478154} {"timestamp_utc": "2026-04-11T23:56:33Z", "mode": "train", "global_step": 1298, "epoch": 0.05213479535687031, "loss": 0.0077, "grad_norm": 7.133460521697998, "learning_rate": 6.06969696969697e-06, "num_tokens": 2930480.0, "completions/mean_length": 34.0, "completions/min_length": 33.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9618321657180786, "rewards/meter/std": 0.022519638761878014, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9618321657180786, "rewards/total_composite/std": 0.022519638761878014, "reward": 0.9618321657180786, "reward_std": 0.022519640624523163, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04984576627612114, "sampling/sampling_logp_difference/max": 0.7256646156311035, "sampling/importance_sampling_ratio/min": 0.4840027689933777, "sampling/importance_sampling_ratio/mean": 1.023314118385315, "sampling/importance_sampling_ratio/max": 1.5166521072387695, "entropy": 0.4743592441082001, "clip_ratio/low_mean": 0.014928699005395174, "clip_ratio/low_min": 0.014928699005395174, "clip_ratio/high_mean": 0.02205882384441793, "clip_ratio/high_max": 0.02205882384441793, "clip_ratio/region_mean": 0.036987522849813104, "reward_total_mean": 0.9618321657180786, "reward_meter_mean": 0.9618321657180786, "reward_meter_std": 0.022519638761878014, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9618321657180786, "reward_total_composite_std": 0.022519638761878014} {"timestamp_utc": "2026-04-11T23:56:38Z", "mode": "train", "global_step": 1299, "epoch": 0.05217496083865526, "loss": -0.0198, "grad_norm": 6.881636142730713, "learning_rate": 6.066666666666667e-06, "num_tokens": 2932528.0, "completions/mean_length": 91.0, "completions/min_length": 86.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.0, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.37875261902809143, "rewards/meter/std": 0.31527000665664673, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.36679619550704956, "rewards/total_composite/std": 0.3127610683441162, "reward": 0.36679619550704956, "reward_std": 0.3127610683441162, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0896112248301506, "sampling/sampling_logp_difference/max": 1.3641595840454102, "sampling/importance_sampling_ratio/min": 0.2555953860282898, "sampling/importance_sampling_ratio/mean": 1.0225510597229004, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7791374623775482, "clip_ratio/low_mean": 0.04831210756674409, "clip_ratio/low_min": 0.04831210756674409, "clip_ratio/high_mean": 0.02266476070508361, "clip_ratio/high_max": 0.02266476070508361, "clip_ratio/region_mean": 0.0709768682718277, "reward_total_mean": 0.36679619550704956, "reward_meter_mean": 0.37875261902809143, "reward_meter_std": 0.31527000665664673, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.36679619550704956, "reward_total_composite_std": 0.3127610683441162} {"timestamp_utc": "2026-04-11T23:56:43Z", "mode": "train", "global_step": 1300, "epoch": 0.05221512632044022, "loss": 0.0288, "grad_norm": 10.192344665527344, "learning_rate": 6.063636363636364e-06, "num_tokens": 2933984.0, "completions/mean_length": 38.0, "completions/min_length": 37.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.0, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.6407955884933472, "rewards/meter/std": 0.46960803866386414, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6407955884933472, "rewards/total_composite/std": 0.46960803866386414, "reward": 0.6407955884933472, "reward_std": 0.46960803866386414, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07035049796104431, "sampling/sampling_logp_difference/max": 1.5472084283828735, "sampling/importance_sampling_ratio/min": 0.21284130215644836, "sampling/importance_sampling_ratio/mean": 1.007232427597046, "sampling/importance_sampling_ratio/max": 1.5791682004928589, "entropy": 0.4784490168094635, "clip_ratio/low_mean": 0.016562293516471982, "clip_ratio/low_min": 0.016562293516471982, "clip_ratio/high_mean": 0.04649715404957533, "clip_ratio/high_max": 0.04649715404957533, "clip_ratio/region_mean": 0.06305944756604731, "reward_total_mean": 0.6407955884933472, "reward_meter_mean": 0.6407955884933472, "reward_meter_std": 0.46960803866386414, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6407955884933472, "reward_total_composite_std": 0.46960803866386414} {"timestamp_utc": "2026-04-11T23:57:39Z", "mode": "eval", "global_step": 1300, "epoch": 0.05221512632044022, "eval_loss": NaN, "eval_runtime": 55.9753, "eval_samples_per_second": 1.858, "eval_steps_per_second": 0.232, "eval_num_tokens": 2933984.0, "eval_completions/mean_length": 170.44230769230768, "eval_completions/min_length": 62.07692307692308, "eval_completions/max_length": 285.9230769230769, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 170.44230769230768, "eval_completions/min_terminated_length": 62.07692307692308, "eval_completions/max_terminated_length": 285.9230769230769, "eval_rewards/meter/mean": 0.5045671004515427, "eval_rewards/meter/std": 0.4269758417056157, "eval_rewards/count_adherence/mean": 0.8285529888593234, "eval_rewards/count_adherence/std": 0.17163905444053504, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/repeat_penalty/mean": 0.8525567375696622, "eval_rewards/repeat_penalty/std": 0.13138744378319153, "eval_rewards/total_composite/mean": 0.36767612856168014, "eval_rewards/total_composite/std": 0.36054489933527434, "eval_reward": 0.36767612856168014, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.02532697569292325, "eval_sampling/sampling_logp_difference/max": 1.0623591863192046, "eval_sampling/importance_sampling_ratio/min": 0.35662598334825957, "eval_sampling/importance_sampling_ratio/mean": 1.0066082569269033, "eval_sampling/importance_sampling_ratio/max": 1.4160877099403968, "eval_entropy": 0.24410619873266953, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.36767612856168014, "eval_reward_meter_mean": 0.5045671004515427, "eval_reward_meter_std": 0.4269758417056157, "eval_reward_count_adherence_mean": 0.8285529888593234, "eval_reward_count_adherence_std": 0.17163905444053504, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_repeat_penalty_mean": 0.8525567375696622, "eval_reward_repeat_penalty_std": 0.13138744378319153, "eval_reward_total_composite_mean": 0.36767612856168014, "eval_reward_total_composite_std": 0.36054489933527434} {"timestamp_utc": "2026-04-11T23:57:47Z", "mode": "train", "global_step": 1301, "epoch": 0.05225529180222517, "loss": 0.0021, "grad_norm": 5.742159843444824, "learning_rate": 6.060606060606061e-06, "num_tokens": 2935839.0, "completions/mean_length": 71.875, "completions/min_length": 67.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.875, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.5558804869651794, "rewards/meter/std": 0.43805626034736633, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5558804869651794, "rewards/total_composite/std": 0.43805626034736633, "reward": 0.5558804869651794, "reward_std": 0.43805623054504395, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.061652250587940216, "sampling/sampling_logp_difference/max": 1.8877463340759277, "sampling/importance_sampling_ratio/min": 0.15141265094280243, "sampling/importance_sampling_ratio/mean": 1.0061253309249878, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47279578633606434, "clip_ratio/low_mean": 0.010581943904981017, "clip_ratio/low_min": 0.010581943904981017, "clip_ratio/high_mean": 0.03823378193192184, "clip_ratio/high_max": 0.03823378193192184, "clip_ratio/region_mean": 0.04881572583690286, "reward_total_mean": 0.5558804869651794, "reward_meter_mean": 0.5558804869651794, "reward_meter_std": 0.43805626034736633, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5558804869651794, "reward_total_composite_std": 0.43805626034736633} {"timestamp_utc": "2026-04-11T23:57:55Z", "mode": "train", "global_step": 1302, "epoch": 0.052295457284010124, "loss": 0.0079, "grad_norm": 1.432262897491455, "learning_rate": 6.057575757575757e-06, "num_tokens": 2939778.0, "completions/mean_length": 322.375, "completions/min_length": 317.0, "completions/max_length": 328.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 322.375, "completions/min_terminated_length": 317.0, "completions/max_terminated_length": 328.0, "rewards/meter/mean": 0.8052475452423096, "rewards/meter/std": 0.0374782457947731, "rewards/count_adherence/mean": 0.5333333611488342, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7666666507720947, "rewards/repeat_penalty/std": 0.06172133609652519, "rewards/total_composite/mean": 0.32962822914123535, "rewards/total_composite/std": 0.0352957658469677, "reward": 0.32962822914123535, "reward_std": 0.0352957658469677, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024004999548196793, "sampling/sampling_logp_difference/max": 1.421916127204895, "sampling/importance_sampling_ratio/min": 0.24125131964683533, "sampling/importance_sampling_ratio/mean": 1.0048707723617554, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1677823355421424, "clip_ratio/low_mean": 0.013534625410102308, "clip_ratio/low_min": 0.013534625410102308, "clip_ratio/high_mean": 0.005509414477273822, "clip_ratio/high_max": 0.005509414477273822, "clip_ratio/region_mean": 0.01904403988737613, "reward_total_mean": 0.32962822914123535, "reward_meter_mean": 0.8052475452423096, "reward_meter_std": 0.0374782457947731, "reward_count_adherence_mean": 0.5333333611488342, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7666666507720947, "reward_repeat_penalty_std": 0.06172133609652519, "reward_total_composite_mean": 0.32962822914123535, "reward_total_composite_std": 0.0352957658469677} {"timestamp_utc": "2026-04-11T23:58:00Z", "mode": "train", "global_step": 1303, "epoch": 0.05233562276579508, "loss": 0.0168, "grad_norm": 3.591283082962036, "learning_rate": 6.0545454545454555e-06, "num_tokens": 2942114.0, "completions/mean_length": 119.0, "completions/min_length": 113.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 119.0, "completions/min_terminated_length": 113.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.8989412188529968, "rewards/meter/std": 0.138026162981987, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7678571343421936, "rewards/repeat_penalty/std": 0.15152288973331451, "rewards/total_composite/mean": 0.6865410804748535, "rewards/total_composite/std": 0.16533121466636658, "reward": 0.6865410804748535, "reward_std": 0.16533119976520538, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027758914977312088, "sampling/sampling_logp_difference/max": 1.322387456893921, "sampling/importance_sampling_ratio/min": 0.26649829745292664, "sampling/importance_sampling_ratio/mean": 1.0041345357894897, "sampling/importance_sampling_ratio/max": 1.522925853729248, "entropy": 0.17016714438796043, "clip_ratio/low_mean": 0.010266162920743227, "clip_ratio/low_min": 0.010266162920743227, "clip_ratio/high_mean": 0.018164411187171936, "clip_ratio/high_max": 0.018164411187171936, "clip_ratio/region_mean": 0.028430574107915163, "reward_total_mean": 0.6865410804748535, "reward_meter_mean": 0.8989412188529968, "reward_meter_std": 0.138026162981987, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7678571343421936, "reward_repeat_penalty_std": 0.15152288973331451, "reward_total_composite_mean": 0.6865410804748535, "reward_total_composite_std": 0.16533121466636658} {"timestamp_utc": "2026-04-11T23:58:08Z", "mode": "train", "global_step": 1304, "epoch": 0.05237578824758003, "loss": -0.0515, "grad_norm": 3.418581962585449, "learning_rate": 6.051515151515152e-06, "num_tokens": 2945734.0, "completions/mean_length": 247.5, "completions/min_length": 232.0, "completions/max_length": 277.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 247.5, "completions/min_terminated_length": 232.0, "completions/max_terminated_length": 277.0, "rewards/meter/mean": 0.9368978142738342, "rewards/meter/std": 0.09892511367797852, "rewards/count_adherence/mean": 0.796875, "rewards/count_adherence/std": 0.06469365209341049, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7849650382995605, "rewards/repeat_penalty/std": 0.07891790568828583, "rewards/total_composite/mean": 0.5886770486831665, "rewards/total_composite/std": 0.11287946254014969, "reward": 0.5886770486831665, "reward_std": 0.11287945508956909, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024336956441402435, "sampling/sampling_logp_difference/max": 1.6146456003189087, "sampling/importance_sampling_ratio/min": 0.19896115362644196, "sampling/importance_sampling_ratio/mean": 1.0041065216064453, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1491470541805029, "clip_ratio/low_mean": 0.00956214708276093, "clip_ratio/low_min": 0.00956214708276093, "clip_ratio/high_mean": 0.011577953351661563, "clip_ratio/high_max": 0.011577953351661563, "clip_ratio/region_mean": 0.021140100434422493, "reward_total_mean": 0.5886770486831665, "reward_meter_mean": 0.9368978142738342, "reward_meter_std": 0.09892511367797852, "reward_count_adherence_mean": 0.796875, "reward_count_adherence_std": 0.06469365209341049, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7849650382995605, "reward_repeat_penalty_std": 0.07891790568828583, "reward_total_composite_mean": 0.5886770486831665, "reward_total_composite_std": 0.11287946254014969} {"timestamp_utc": "2026-04-11T23:58:14Z", "mode": "train", "global_step": 1305, "epoch": 0.052415953729364986, "loss": 0.0351, "grad_norm": 3.999876022338867, "learning_rate": 6.048484848484849e-06, "num_tokens": 2948213.0, "completions/mean_length": 132.875, "completions/min_length": 125.0, "completions/max_length": 138.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.875, "completions/min_terminated_length": 125.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.45057880878448486, "rewards/meter/std": 0.39947640895843506, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.910714328289032, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.4297538995742798, "rewards/total_composite/std": 0.3954527974128723, "reward": 0.4297538995742798, "reward_std": 0.3954527974128723, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03056526370346546, "sampling/sampling_logp_difference/max": 1.2996227741241455, "sampling/importance_sampling_ratio/min": 0.2726346254348755, "sampling/importance_sampling_ratio/mean": 1.0062963962554932, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15318176615983248, "clip_ratio/low_mean": 0.006469378364272416, "clip_ratio/low_min": 0.006469378364272416, "clip_ratio/high_mean": 0.02036819839850068, "clip_ratio/high_max": 0.02036819839850068, "clip_ratio/region_mean": 0.026837576762773097, "reward_total_mean": 0.4297538995742798, "reward_meter_mean": 0.45057880878448486, "reward_meter_std": 0.39947640895843506, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.910714328289032, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.4297538995742798, "reward_total_composite_std": 0.3954527974128723} {"timestamp_utc": "2026-04-11T23:58:19Z", "mode": "train", "global_step": 1306, "epoch": 0.05245611921114994, "loss": 0.0182, "grad_norm": 5.305455207824707, "learning_rate": 6.0454545454545456e-06, "num_tokens": 2949699.0, "completions/mean_length": 38.75, "completions/min_length": 33.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.75, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.9870239496231079, "rewards/meter/std": 0.016622211784124374, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9870239496231079, "rewards/total_composite/std": 0.016622211784124374, "reward": 0.9870239496231079, "reward_std": 0.016622209921479225, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05584869906306267, "sampling/sampling_logp_difference/max": 1.2581281661987305, "sampling/importance_sampling_ratio/min": 0.2841854691505432, "sampling/importance_sampling_ratio/mean": 1.015398621559143, "sampling/importance_sampling_ratio/max": 1.6217622756958008, "entropy": 0.4362948350608349, "clip_ratio/low_mean": 0.0031250000465661287, "clip_ratio/low_min": 0.0031250000465661287, "clip_ratio/high_mean": 0.031160252634435892, "clip_ratio/high_max": 0.031160252634435892, "clip_ratio/region_mean": 0.03428525268100202, "reward_total_mean": 0.9870239496231079, "reward_meter_mean": 0.9870239496231079, "reward_meter_std": 0.016622211784124374, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9870239496231079, "reward_total_composite_std": 0.016622211784124374} {"timestamp_utc": "2026-04-11T23:58:24Z", "mode": "train", "global_step": 1307, "epoch": 0.052496284692934894, "loss": 0.0252, "grad_norm": 4.432555198669434, "learning_rate": 6.042424242424243e-06, "num_tokens": 2951929.0, "completions/mean_length": 96.75, "completions/min_length": 86.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.75, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.6255627274513245, "rewards/meter/std": 0.33046483993530273, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9249999523162842, "rewards/repeat_penalty/std": 0.14880475401878357, "rewards/total_composite/mean": 0.5558610558509827, "rewards/total_composite/std": 0.2833639085292816, "reward": 0.5558610558509827, "reward_std": 0.28336387872695923, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.048519354313611984, "sampling/sampling_logp_difference/max": 1.356567621231079, "sampling/importance_sampling_ratio/min": 0.25754326581954956, "sampling/importance_sampling_ratio/mean": 1.008318543434143, "sampling/importance_sampling_ratio/max": 1.8040260076522827, "entropy": 0.36078083515167236, "clip_ratio/low_mean": 0.012753612943924963, "clip_ratio/low_min": 0.012753612943924963, "clip_ratio/high_mean": 0.023235379019752145, "clip_ratio/high_max": 0.023235379019752145, "clip_ratio/region_mean": 0.03598899196367711, "reward_total_mean": 0.5558610558509827, "reward_meter_mean": 0.6255627274513245, "reward_meter_std": 0.33046483993530273, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9249999523162842, "reward_repeat_penalty_std": 0.14880475401878357, "reward_total_composite_mean": 0.5558610558509827, "reward_total_composite_std": 0.2833639085292816} {"timestamp_utc": "2026-04-11T23:58:28Z", "mode": "train", "global_step": 1308, "epoch": 0.05253645017471985, "loss": 0.0072, "grad_norm": 6.026689529418945, "learning_rate": 6.039393939393939e-06, "num_tokens": 2953756.0, "completions/mean_length": 64.375, "completions/min_length": 63.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.375, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9559545516967773, "rewards/meter/std": 0.06238928064703941, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9559545516967773, "rewards/total_composite/std": 0.06238928064703941, "reward": 0.9559545516967773, "reward_std": 0.062389273196458817, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031952135264873505, "sampling/sampling_logp_difference/max": 2.3071088790893555, "sampling/importance_sampling_ratio/min": 0.09954863786697388, "sampling/importance_sampling_ratio/mean": 1.000700831413269, "sampling/importance_sampling_ratio/max": 1.5961639881134033, "entropy": 0.11791138723492622, "clip_ratio/low_mean": 0.003846153849735856, "clip_ratio/low_min": 0.003846153849735856, "clip_ratio/high_mean": 0.021340470295399427, "clip_ratio/high_max": 0.021340470295399427, "clip_ratio/region_mean": 0.025186624145135283, "reward_total_mean": 0.9559545516967773, "reward_meter_mean": 0.9559545516967773, "reward_meter_std": 0.06238928064703941, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9559545516967773, "reward_total_composite_std": 0.06238928064703941} {"timestamp_utc": "2026-04-11T23:58:32Z", "mode": "train", "global_step": 1309, "epoch": 0.0525766156565048, "loss": 0.0141, "grad_norm": 6.288991928100586, "learning_rate": 6.0363636363636365e-06, "num_tokens": 2955303.0, "completions/mean_length": 34.375, "completions/min_length": 32.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.375, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.8419111371040344, "rewards/meter/std": 0.3378627300262451, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8419111371040344, "rewards/total_composite/std": 0.3378627300262451, "reward": 0.8419111371040344, "reward_std": 0.33786270022392273, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03979434072971344, "sampling/sampling_logp_difference/max": 0.6886262893676758, "sampling/importance_sampling_ratio/min": 0.5022655725479126, "sampling/importance_sampling_ratio/mean": 1.0118752717971802, "sampling/importance_sampling_ratio/max": 1.8161826133728027, "entropy": 0.3678742181509733, "clip_ratio/low_mean": 0.0035714285913854837, "clip_ratio/low_min": 0.0035714285913854837, "clip_ratio/high_mean": 0.030007666442543268, "clip_ratio/high_max": 0.030007666442543268, "clip_ratio/region_mean": 0.03357909503392875, "reward_total_mean": 0.8419111371040344, "reward_meter_mean": 0.8419111371040344, "reward_meter_std": 0.3378627300262451, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8419111371040344, "reward_total_composite_std": 0.3378627300262451} {"timestamp_utc": "2026-04-11T23:58:37Z", "mode": "train", "global_step": 1310, "epoch": 0.052616781138289756, "loss": -0.0014, "grad_norm": 4.350625514984131, "learning_rate": 6.033333333333335e-06, "num_tokens": 2957280.0, "completions/mean_length": 70.125, "completions/min_length": 69.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.125, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9980206489562988, "rewards/meter/std": 0.00018548252410255373, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980206489562988, "rewards/total_composite/std": 0.00018548252410255373, "reward": 0.9980206489562988, "reward_std": 0.00018549479136709124, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022730743512511253, "sampling/sampling_logp_difference/max": 1.1889805793762207, "sampling/importance_sampling_ratio/min": 0.3045315444469452, "sampling/importance_sampling_ratio/mean": 1.0046977996826172, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10299927368760109, "clip_ratio/low_mean": 0.010817805537953973, "clip_ratio/low_min": 0.010817805537953973, "clip_ratio/high_mean": 0.0035462776431813836, "clip_ratio/high_max": 0.0035462776431813836, "clip_ratio/region_mean": 0.014364083181135356, "reward_total_mean": 0.9980206489562988, "reward_meter_mean": 0.9980206489562988, "reward_meter_std": 0.00018548252410255373, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9980206489562988, "reward_total_composite_std": 0.00018548252410255373} {"timestamp_utc": "2026-04-11T23:58:42Z", "mode": "train", "global_step": 1311, "epoch": 0.05265694662007471, "loss": 0.0065, "grad_norm": 2.434211015701294, "learning_rate": 6.030303030303031e-06, "num_tokens": 2959801.0, "completions/mean_length": 140.125, "completions/min_length": 139.0, "completions/max_length": 141.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 140.125, "completions/min_terminated_length": 139.0, "completions/max_terminated_length": 141.0, "rewards/meter/mean": 0.9983865022659302, "rewards/meter/std": 2.5610979719203897e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.09155284613370895, "rewards/total_composite/mean": 0.7309621572494507, "rewards/total_composite/std": 0.09140855073928833, "reward": 0.7309621572494507, "reward_std": 0.09140852838754654, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021905627101659775, "sampling/sampling_logp_difference/max": 1.7227790355682373, "sampling/importance_sampling_ratio/min": 0.17856921255588531, "sampling/importance_sampling_ratio/mean": 1.0040990114212036, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11066047195345163, "clip_ratio/low_mean": 0.01160750730196014, "clip_ratio/low_min": 0.01160750730196014, "clip_ratio/high_mean": 0.0035714286495931447, "clip_ratio/high_max": 0.0035714286495931447, "clip_ratio/region_mean": 0.015178935951553285, "reward_total_mean": 0.7309621572494507, "reward_meter_mean": 0.9983865022659302, "reward_meter_std": 2.5610979719203897e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.09155284613370895, "reward_total_composite_mean": 0.7309621572494507, "reward_total_composite_std": 0.09140855073928833} {"timestamp_utc": "2026-04-11T23:58:46Z", "mode": "train", "global_step": 1312, "epoch": 0.052697112101859664, "loss": -0.0003, "grad_norm": 5.163435459136963, "learning_rate": 6.027272727272728e-06, "num_tokens": 2961576.0, "completions/mean_length": 59.875, "completions/min_length": 59.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.875, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.8866318464279175, "rewards/meter/std": 0.18304355442523956, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8866318464279175, "rewards/total_composite/std": 0.18304355442523956, "reward": 0.8866318464279175, "reward_std": 0.18304355442523956, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02904520370066166, "sampling/sampling_logp_difference/max": 0.9866929054260254, "sampling/importance_sampling_ratio/min": 0.3728075623512268, "sampling/importance_sampling_ratio/mean": 1.0048235654830933, "sampling/importance_sampling_ratio/max": 1.9061437845230103, "entropy": 0.16338308528065681, "clip_ratio/low_mean": 0.006250000325962901, "clip_ratio/low_min": 0.006250000325962901, "clip_ratio/high_mean": 0.016737288795411587, "clip_ratio/high_max": 0.016737288795411587, "clip_ratio/region_mean": 0.022987289121374488, "reward_total_mean": 0.8866318464279175, "reward_meter_mean": 0.8866318464279175, "reward_meter_std": 0.18304355442523956, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8866318464279175, "reward_total_composite_std": 0.18304355442523956} {"timestamp_utc": "2026-04-11T23:58:52Z", "mode": "train", "global_step": 1313, "epoch": 0.05273727758364462, "loss": 0.0116, "grad_norm": 3.3765552043914795, "learning_rate": 6.024242424242425e-06, "num_tokens": 2964564.0, "completions/mean_length": 172.5, "completions/min_length": 168.0, "completions/max_length": 179.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 172.5, "completions/min_terminated_length": 168.0, "completions/max_terminated_length": 179.0, "rewards/meter/mean": 0.7292158007621765, "rewards/meter/std": 0.2301097810268402, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.819444477558136, "rewards/repeat_penalty/std": 0.08266931027173996, "rewards/total_composite/mean": 0.49107223749160767, "rewards/total_composite/std": 0.1431042104959488, "reward": 0.49107223749160767, "reward_std": 0.1431041955947876, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.040278006345033646, "sampling/sampling_logp_difference/max": 1.5737600326538086, "sampling/importance_sampling_ratio/min": 0.20726439356803894, "sampling/importance_sampling_ratio/mean": 1.005394697189331, "sampling/importance_sampling_ratio/max": 1.8441625833511353, "entropy": 0.2585892491042614, "clip_ratio/low_mean": 0.010145591222681105, "clip_ratio/low_min": 0.010145591222681105, "clip_ratio/high_mean": 0.01937754324171692, "clip_ratio/high_max": 0.01937754324171692, "clip_ratio/region_mean": 0.029523134464398026, "reward_total_mean": 0.49107223749160767, "reward_meter_mean": 0.7292158007621765, "reward_meter_std": 0.2301097810268402, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.819444477558136, "reward_repeat_penalty_std": 0.08266931027173996, "reward_total_composite_mean": 0.49107223749160767, "reward_total_composite_std": 0.1431042104959488} {"timestamp_utc": "2026-04-11T23:58:57Z", "mode": "train", "global_step": 1314, "epoch": 0.05277744306542957, "loss": -0.0091, "grad_norm": 5.761319637298584, "learning_rate": 6.021212121212122e-06, "num_tokens": 2966484.0, "completions/mean_length": 74.0, "completions/min_length": 70.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.0, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.7603874802589417, "rewards/meter/std": 0.4266778230667114, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7603874802589417, "rewards/total_composite/std": 0.4266778230667114, "reward": 0.7603874802589417, "reward_std": 0.4266778230667114, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.042963117361068726, "sampling/sampling_logp_difference/max": 1.180001974105835, "sampling/importance_sampling_ratio/min": 0.3072781264781952, "sampling/importance_sampling_ratio/mean": 0.9964911341667175, "sampling/importance_sampling_ratio/max": 1.84821355342865, "entropy": 0.24265113845467567, "clip_ratio/low_mean": 0.005260617821477354, "clip_ratio/low_min": 0.005260617821477354, "clip_ratio/high_mean": 0.043520811945199966, "clip_ratio/high_max": 0.043520811945199966, "clip_ratio/region_mean": 0.04878142976667732, "reward_total_mean": 0.7603874802589417, "reward_meter_mean": 0.7603874802589417, "reward_meter_std": 0.4266778230667114, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7603874802589417, "reward_total_composite_std": 0.4266778230667114} {"timestamp_utc": "2026-04-11T23:59:02Z", "mode": "train", "global_step": 1315, "epoch": 0.052817608547214526, "loss": 0.0157, "grad_norm": 12.346835136413574, "learning_rate": 6.018181818181818e-06, "num_tokens": 2968868.0, "completions/mean_length": 109.0, "completions/min_length": 106.0, "completions/max_length": 115.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 109.0, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 115.0, "rewards/meter/mean": 0.9957406520843506, "rewards/meter/std": 0.0016969876596704125, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957406520843506, "rewards/total_composite/std": 0.0016969876596704125, "reward": 0.9957406520843506, "reward_std": 0.0016970024444162846, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.050970617681741714, "sampling/sampling_logp_difference/max": 1.3137094974517822, "sampling/importance_sampling_ratio/min": 0.26882100105285645, "sampling/importance_sampling_ratio/mean": 1.0082180500030518, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3228848744183779, "clip_ratio/low_mean": 0.013646618113853037, "clip_ratio/low_min": 0.013646618113853037, "clip_ratio/high_mean": 0.018337647430598736, "clip_ratio/high_max": 0.018337647430598736, "clip_ratio/region_mean": 0.03198426554445177, "reward_total_mean": 0.9957406520843506, "reward_meter_mean": 0.9957406520843506, "reward_meter_std": 0.0016969876596704125, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9957406520843506, "reward_total_composite_std": 0.0016969876596704125} {"timestamp_utc": "2026-04-11T23:59:07Z", "mode": "train", "global_step": 1316, "epoch": 0.05285777402899948, "loss": 0.0078, "grad_norm": 5.835085868835449, "learning_rate": 6.015151515151516e-06, "num_tokens": 2970696.0, "completions/mean_length": 63.5, "completions/min_length": 62.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.5, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.5161525011062622, "rewards/meter/std": 0.3657090961933136, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4984230399131775, "rewards/total_composite/std": 0.3891359865665436, "reward": 0.4984230399131775, "reward_std": 0.3891359865665436, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03714925795793533, "sampling/sampling_logp_difference/max": 2.0932350158691406, "sampling/importance_sampling_ratio/min": 0.12328764796257019, "sampling/importance_sampling_ratio/mean": 1.006494402885437, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.21594705432653427, "clip_ratio/low_mean": 0.025765649508684874, "clip_ratio/low_min": 0.025765649508684874, "clip_ratio/high_mean": 0.01568986615166068, "clip_ratio/high_max": 0.01568986615166068, "clip_ratio/region_mean": 0.041455515660345554, "reward_total_mean": 0.4984230399131775, "reward_meter_mean": 0.5161525011062622, "reward_meter_std": 0.3657090961933136, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.4984230399131775, "reward_total_composite_std": 0.3891359865665436} {"timestamp_utc": "2026-04-11T23:59:12Z", "mode": "train", "global_step": 1317, "epoch": 0.052897939510784434, "loss": 0.0052, "grad_norm": 3.077263832092285, "learning_rate": 6.012121212121213e-06, "num_tokens": 2972590.0, "completions/mean_length": 75.75, "completions/min_length": 74.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.75, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.2975860834121704, "rewards/meter/std": 0.04687010124325752, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.26845037937164307, "rewards/total_composite/std": 0.04051537066698074, "reward": 0.26845037937164307, "reward_std": 0.040515363216400146, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02295822836458683, "sampling/sampling_logp_difference/max": 0.9676837921142578, "sampling/importance_sampling_ratio/min": 0.37996208667755127, "sampling/importance_sampling_ratio/mean": 0.999930739402771, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09577750787138939, "clip_ratio/low_mean": 0.008204055950045586, "clip_ratio/low_min": 0.008204055950045586, "clip_ratio/high_mean": 0.006602564244531095, "clip_ratio/high_max": 0.006602564244531095, "clip_ratio/region_mean": 0.01480662019457668, "reward_total_mean": 0.26845037937164307, "reward_meter_mean": 0.2975860834121704, "reward_meter_std": 0.04687010124325752, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_total_composite_mean": 0.26845037937164307, "reward_total_composite_std": 0.04051537066698074} {"timestamp_utc": "2026-04-11T23:59:16Z", "mode": "train", "global_step": 1318, "epoch": 0.05293810499256939, "loss": 0.0333, "grad_norm": 6.030852317810059, "learning_rate": 6.00909090909091e-06, "num_tokens": 2974432.0, "completions/mean_length": 51.25, "completions/min_length": 48.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.25, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.813197135925293, "rewards/meter/std": 0.24947111308574677, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.813197135925293, "rewards/total_composite/std": 0.24947111308574677, "reward": 0.813197135925293, "reward_std": 0.24947108328342438, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.047421231865882874, "sampling/sampling_logp_difference/max": 1.33819580078125, "sampling/importance_sampling_ratio/min": 0.26231852173805237, "sampling/importance_sampling_ratio/mean": 1.0154815912246704, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.268137663602829, "clip_ratio/low_mean": 0.009302935097366571, "clip_ratio/low_min": 0.009302935097366571, "clip_ratio/high_mean": 0.012667326722294092, "clip_ratio/high_max": 0.012667326722294092, "clip_ratio/region_mean": 0.021970261819660664, "reward_total_mean": 0.813197135925293, "reward_meter_mean": 0.813197135925293, "reward_meter_std": 0.24947111308574677, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.813197135925293, "reward_total_composite_std": 0.24947111308574677} {"timestamp_utc": "2026-04-11T23:59:23Z", "mode": "train", "global_step": 1319, "epoch": 0.05297827047435434, "loss": 0.0064, "grad_norm": 2.1162424087524414, "learning_rate": 6.0060606060606065e-06, "num_tokens": 2978169.0, "completions/mean_length": 256.125, "completions/min_length": 232.0, "completions/max_length": 267.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 256.125, "completions/min_terminated_length": 232.0, "completions/max_terminated_length": 267.0, "rewards/meter/mean": 0.9216792583465576, "rewards/meter/std": 0.17728787660598755, "rewards/count_adherence/mean": 0.859375, "rewards/count_adherence/std": 0.04419417306780815, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7475961446762085, "rewards/repeat_penalty/std": 0.0792488381266594, "rewards/total_composite/mean": 0.586740255355835, "rewards/total_composite/std": 0.11694769561290741, "reward": 0.586740255355835, "reward_std": 0.11694768071174622, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026936307549476624, "sampling/sampling_logp_difference/max": 2.076201915740967, "sampling/importance_sampling_ratio/min": 0.12540560960769653, "sampling/importance_sampling_ratio/mean": 1.0010830163955688, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.14165427815169096, "clip_ratio/low_mean": 0.006019041407853365, "clip_ratio/low_min": 0.006019041407853365, "clip_ratio/high_mean": 0.012993455864489079, "clip_ratio/high_max": 0.012993455864489079, "clip_ratio/region_mean": 0.019012497272342443, "reward_total_mean": 0.586740255355835, "reward_meter_mean": 0.9216792583465576, "reward_meter_std": 0.17728787660598755, "reward_count_adherence_mean": 0.859375, "reward_count_adherence_std": 0.04419417306780815, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7475961446762085, "reward_repeat_penalty_std": 0.0792488381266594, "reward_total_composite_mean": 0.586740255355835, "reward_total_composite_std": 0.11694769561290741} {"timestamp_utc": "2026-04-11T23:59:28Z", "mode": "train", "global_step": 1320, "epoch": 0.053018435956139295, "loss": 0.0292, "grad_norm": 6.174464225769043, "learning_rate": 6.003030303030304e-06, "num_tokens": 2980004.0, "completions/mean_length": 63.375, "completions/min_length": 59.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.375, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.6976158022880554, "rewards/meter/std": 0.3385031819343567, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6976158022880554, "rewards/total_composite/std": 0.3385031819343567, "reward": 0.6976158022880554, "reward_std": 0.3385031521320343, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06314612179994583, "sampling/sampling_logp_difference/max": 1.4771289825439453, "sampling/importance_sampling_ratio/min": 0.22829218208789825, "sampling/importance_sampling_ratio/mean": 1.0181591510772705, "sampling/importance_sampling_ratio/max": 1.8571363687515259, "entropy": 0.4745039828121662, "clip_ratio/low_mean": 0.007752403849735856, "clip_ratio/low_min": 0.007752403849735856, "clip_ratio/high_mean": 0.03416440007276833, "clip_ratio/high_max": 0.03416440007276833, "clip_ratio/region_mean": 0.04191680392250419, "reward_total_mean": 0.6976158022880554, "reward_meter_mean": 0.6976158022880554, "reward_meter_std": 0.3385031819343567, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6976158022880554, "reward_total_composite_std": 0.3385031819343567} {"timestamp_utc": "2026-04-11T23:59:32Z", "mode": "train", "global_step": 1321, "epoch": 0.05305860143792425, "loss": 0.0145, "grad_norm": 4.180057525634766, "learning_rate": 6e-06, "num_tokens": 2981843.0, "completions/mean_length": 72.875, "completions/min_length": 68.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.875, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.8855767846107483, "rewards/meter/std": 0.09589327871799469, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8855767846107483, "rewards/total_composite/std": 0.09589327871799469, "reward": 0.8855767846107483, "reward_std": 0.09589327871799469, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03412603959441185, "sampling/sampling_logp_difference/max": 1.985456943511963, "sampling/importance_sampling_ratio/min": 0.13731785118579865, "sampling/importance_sampling_ratio/mean": 0.9989095330238342, "sampling/importance_sampling_ratio/max": 1.6174578666687012, "entropy": 0.1369091346859932, "clip_ratio/low_mean": 0.008787594037130475, "clip_ratio/low_min": 0.008787594037130475, "clip_ratio/high_mean": 0.026141698006540537, "clip_ratio/high_max": 0.026141698006540537, "clip_ratio/region_mean": 0.03492929204367101, "reward_total_mean": 0.8855767846107483, "reward_meter_mean": 0.8855767846107483, "reward_meter_std": 0.09589327871799469, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8855767846107483, "reward_total_composite_std": 0.09589327871799469} {"timestamp_utc": "2026-04-11T23:59:36Z", "mode": "train", "global_step": 1322, "epoch": 0.0530987669197092, "loss": -0.0035, "grad_norm": 3.336163282394409, "learning_rate": 5.996969696969697e-06, "num_tokens": 2983446.0, "completions/mean_length": 41.375, "completions/min_length": 40.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.375, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.9977743029594421, "rewards/meter/std": 0.0008235168061219156, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977743029594421, "rewards/total_composite/std": 0.0008235168061219156, "reward": 0.9977743029594421, "reward_std": 0.0008235144196078181, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.033282674849033356, "sampling/sampling_logp_difference/max": 1.2528910636901855, "sampling/importance_sampling_ratio/min": 0.2856776714324951, "sampling/importance_sampling_ratio/mean": 1.0045777559280396, "sampling/importance_sampling_ratio/max": 1.5313419103622437, "entropy": 0.23666546866297722, "clip_ratio/low_mean": 0.01510935788974166, "clip_ratio/low_min": 0.01510935788974166, "clip_ratio/high_mean": 0.009073751280084252, "clip_ratio/high_max": 0.009073751280084252, "clip_ratio/region_mean": 0.02418310916982591, "reward_total_mean": 0.9977743029594421, "reward_meter_mean": 0.9977743029594421, "reward_meter_std": 0.0008235168061219156, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9977743029594421, "reward_total_composite_std": 0.0008235168061219156} {"timestamp_utc": "2026-04-11T23:59:41Z", "mode": "train", "global_step": 1323, "epoch": 0.05313893240149416, "loss": 0.0011, "grad_norm": 7.640705108642578, "learning_rate": 5.993939393939394e-06, "num_tokens": 2985219.0, "completions/mean_length": 62.625, "completions/min_length": 52.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.625, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.4877135455608368, "rewards/meter/std": 0.3409081995487213, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4877135455608368, "rewards/total_composite/std": 0.3409081995487213, "reward": 0.4877135455608368, "reward_std": 0.3409081995487213, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08788689225912094, "sampling/sampling_logp_difference/max": 1.4542689323425293, "sampling/importance_sampling_ratio/min": 0.23357106745243073, "sampling/importance_sampling_ratio/mean": 1.0144366025924683, "sampling/importance_sampling_ratio/max": 1.7955073118209839, "entropy": 0.6359868273139, "clip_ratio/low_mean": 0.025560760172083974, "clip_ratio/low_min": 0.025560760172083974, "clip_ratio/high_mean": 0.023659866768866777, "clip_ratio/high_max": 0.023659866768866777, "clip_ratio/region_mean": 0.04922062694095075, "reward_total_mean": 0.4877135455608368, "reward_meter_mean": 0.4877135455608368, "reward_meter_std": 0.3409081995487213, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.4877135455608368, "reward_total_composite_std": 0.3409081995487213} {"timestamp_utc": "2026-04-11T23:59:45Z", "mode": "train", "global_step": 1324, "epoch": 0.05317909788327911, "loss": 0.0016, "grad_norm": 6.885532855987549, "learning_rate": 5.990909090909092e-06, "num_tokens": 2986901.0, "completions/mean_length": 41.25, "completions/min_length": 39.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.25, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.8787177801132202, "rewards/meter/std": 0.3274337947368622, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8787177801132202, "rewards/total_composite/std": 0.3274337947368622, "reward": 0.8787177801132202, "reward_std": 0.3274337649345398, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03787970542907715, "sampling/sampling_logp_difference/max": 0.6567374467849731, "sampling/importance_sampling_ratio/min": 0.5185403823852539, "sampling/importance_sampling_ratio/mean": 1.0028668642044067, "sampling/importance_sampling_ratio/max": 1.5247844457626343, "entropy": 0.2660053391009569, "clip_ratio/low_mean": 0.006097560748457909, "clip_ratio/low_min": 0.006097560748457909, "clip_ratio/high_mean": 0.023976000724360347, "clip_ratio/high_max": 0.023976000724360347, "clip_ratio/region_mean": 0.030073561472818255, "reward_total_mean": 0.8787177801132202, "reward_meter_mean": 0.8787177801132202, "reward_meter_std": 0.3274337947368622, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8787177801132202, "reward_total_composite_std": 0.3274337947368622} {"timestamp_utc": "2026-04-11T23:59:50Z", "mode": "train", "global_step": 1325, "epoch": 0.053219263365064065, "loss": -0.0087, "grad_norm": 3.35534405708313, "learning_rate": 5.987878787878788e-06, "num_tokens": 2988985.0, "completions/mean_length": 108.5, "completions/min_length": 104.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.5, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.7801897525787354, "rewards/meter/std": 0.25841277837753296, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.824999988079071, "rewards/repeat_penalty/std": 0.12817399203777313, "rewards/total_composite/mean": 0.6235833168029785, "rewards/total_composite/std": 0.18524423241615295, "reward": 0.6235833168029785, "reward_std": 0.18524421751499176, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03167083114385605, "sampling/sampling_logp_difference/max": 1.2086710929870605, "sampling/importance_sampling_ratio/min": 0.29859381914138794, "sampling/importance_sampling_ratio/mean": 1.0031850337982178, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15250306809321046, "clip_ratio/low_mean": 0.014129945659078658, "clip_ratio/low_min": 0.014129945659078658, "clip_ratio/high_mean": 0.017529349657706916, "clip_ratio/high_max": 0.017529349657706916, "clip_ratio/region_mean": 0.031659295316785574, "reward_total_mean": 0.6235833168029785, "reward_meter_mean": 0.7801897525787354, "reward_meter_std": 0.25841277837753296, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.824999988079071, "reward_repeat_penalty_std": 0.12817399203777313, "reward_total_composite_mean": 0.6235833168029785, "reward_total_composite_std": 0.18524423241615295} {"timestamp_utc": "2026-04-11T23:59:58Z", "mode": "train", "global_step": 1326, "epoch": 0.05325942884684902, "loss": 0.0087, "grad_norm": 2.242483139038086, "learning_rate": 5.984848484848486e-06, "num_tokens": 2992787.0, "completions/mean_length": 258.25, "completions/min_length": 245.0, "completions/max_length": 268.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 258.25, "completions/min_terminated_length": 245.0, "completions/max_terminated_length": 268.0, "rewards/meter/mean": 0.9577704668045044, "rewards/meter/std": 0.11232133209705353, "rewards/count_adherence/mean": 0.5833333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.807692289352417, "rewards/repeat_penalty/std": 0.082234226167202, "rewards/total_composite/mean": 0.4503657817840576, "rewards/total_composite/std": 0.06605512648820877, "reward": 0.4503657817840576, "reward_std": 0.06605512648820877, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02666037157177925, "sampling/sampling_logp_difference/max": 1.849881649017334, "sampling/importance_sampling_ratio/min": 0.15725576877593994, "sampling/importance_sampling_ratio/mean": 1.0053751468658447, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.151836016215384, "clip_ratio/low_mean": 0.01110323509783484, "clip_ratio/low_min": 0.01110323509783484, "clip_ratio/high_mean": 0.014489957015030086, "clip_ratio/high_max": 0.014489957015030086, "clip_ratio/region_mean": 0.025593192112864926, "reward_total_mean": 0.4503657817840576, "reward_meter_mean": 0.9577704668045044, "reward_meter_std": 0.11232133209705353, "reward_count_adherence_mean": 0.5833333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.807692289352417, "reward_repeat_penalty_std": 0.082234226167202, "reward_total_composite_mean": 0.4503657817840576, "reward_total_composite_std": 0.06605512648820877} {"timestamp_utc": "2026-04-12T00:00:03Z", "mode": "train", "global_step": 1327, "epoch": 0.05329959432863397, "loss": -0.0023, "grad_norm": 2.6014902591705322, "learning_rate": 5.981818181818182e-06, "num_tokens": 2995231.0, "completions/mean_length": 139.5, "completions/min_length": 133.0, "completions/max_length": 143.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 139.5, "completions/min_terminated_length": 133.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.9958108067512512, "rewards/meter/std": 0.0007901726639829576, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8928571343421936, "rewards/repeat_penalty/std": 0.10101525485515594, "rewards/total_composite/mean": 0.8890672922134399, "rewards/total_composite/std": 0.1000855341553688, "reward": 0.8890672922134399, "reward_std": 0.10008554905653, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03180097043514252, "sampling/sampling_logp_difference/max": 1.4053449630737305, "sampling/importance_sampling_ratio/min": 0.24528244137763977, "sampling/importance_sampling_ratio/mean": 1.0085175037384033, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.22298285737633705, "clip_ratio/low_mean": 0.008893042104318738, "clip_ratio/low_min": 0.008893042104318738, "clip_ratio/high_mean": 0.002672330185305327, "clip_ratio/high_max": 0.002672330185305327, "clip_ratio/region_mean": 0.011565372289624065, "reward_total_mean": 0.8890672922134399, "reward_meter_mean": 0.9958108067512512, "reward_meter_std": 0.0007901726639829576, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8928571343421936, "reward_repeat_penalty_std": 0.10101525485515594, "reward_total_composite_mean": 0.8890672922134399, "reward_total_composite_std": 0.1000855341553688} {"timestamp_utc": "2026-04-12T00:00:08Z", "mode": "train", "global_step": 1328, "epoch": 0.05333975981041893, "loss": 0.0114, "grad_norm": 9.022930145263672, "learning_rate": 5.978787878787879e-06, "num_tokens": 2997079.0, "completions/mean_length": 61.0, "completions/min_length": 60.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.8425968885421753, "rewards/meter/std": 0.31753626465797424, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8425968885421753, "rewards/total_composite/std": 0.31753626465797424, "reward": 0.8425968885421753, "reward_std": 0.31753626465797424, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04912829399108887, "sampling/sampling_logp_difference/max": 1.3530278205871582, "sampling/importance_sampling_ratio/min": 0.2584564983844757, "sampling/importance_sampling_ratio/mean": 1.0045350790023804, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17815150693058968, "clip_ratio/low_mean": 0.01024590153247118, "clip_ratio/low_min": 0.01024590153247118, "clip_ratio/high_mean": 0.024630822706967592, "clip_ratio/high_max": 0.024630822706967592, "clip_ratio/region_mean": 0.03487672423943877, "reward_total_mean": 0.8425968885421753, "reward_meter_mean": 0.8425968885421753, "reward_meter_std": 0.31753626465797424, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8425968885421753, "reward_total_composite_std": 0.31753626465797424} {"timestamp_utc": "2026-04-12T00:00:13Z", "mode": "train", "global_step": 1329, "epoch": 0.05337992529220388, "loss": 0.0153, "grad_norm": 3.7103993892669678, "learning_rate": 5.975757575757576e-06, "num_tokens": 2999253.0, "completions/mean_length": 89.75, "completions/min_length": 87.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.75, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.7116844058036804, "rewards/meter/std": 0.3912390172481537, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8250000476837158, "rewards/repeat_penalty/std": 0.12817397713661194, "rewards/total_composite/mean": 0.5941954255104065, "rewards/total_composite/std": 0.35485145449638367, "reward": 0.5941954255104065, "reward_std": 0.35485145449638367, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026401184499263763, "sampling/sampling_logp_difference/max": 0.8573803901672363, "sampling/importance_sampling_ratio/min": 0.4242720603942871, "sampling/importance_sampling_ratio/mean": 1.0010446310043335, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1314184283837676, "clip_ratio/low_mean": 0.006886049755848944, "clip_ratio/low_min": 0.006886049755848944, "clip_ratio/high_mean": 0.02516206307336688, "clip_ratio/high_max": 0.02516206307336688, "clip_ratio/region_mean": 0.032048112829215825, "reward_total_mean": 0.5941954255104065, "reward_meter_mean": 0.7116844058036804, "reward_meter_std": 0.3912390172481537, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8250000476837158, "reward_repeat_penalty_std": 0.12817397713661194, "reward_total_composite_mean": 0.5941954255104065, "reward_total_composite_std": 0.35485145449638367} {"timestamp_utc": "2026-04-12T00:00:17Z", "mode": "train", "global_step": 1330, "epoch": 0.053420090773988835, "loss": 0.0134, "grad_norm": 6.578309535980225, "learning_rate": 5.972727272727274e-06, "num_tokens": 3001055.0, "completions/mean_length": 70.25, "completions/min_length": 68.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.25, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.8781044483184814, "rewards/meter/std": 0.32534006237983704, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8781044483184814, "rewards/total_composite/std": 0.32534006237983704, "reward": 0.8781044483184814, "reward_std": 0.32534006237983704, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04965127259492874, "sampling/sampling_logp_difference/max": 1.5530381202697754, "sampling/importance_sampling_ratio/min": 0.21160411834716797, "sampling/importance_sampling_ratio/mean": 1.002303957939148, "sampling/importance_sampling_ratio/max": 1.5147264003753662, "entropy": 0.29124583303928375, "clip_ratio/low_mean": 0.0052083334885537624, "clip_ratio/low_min": 0.0052083334885537624, "clip_ratio/high_mean": 0.025107774534262717, "clip_ratio/high_max": 0.025107774534262717, "clip_ratio/region_mean": 0.03031610802281648, "reward_total_mean": 0.8781044483184814, "reward_meter_mean": 0.8781044483184814, "reward_meter_std": 0.32534006237983704, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8781044483184814, "reward_total_composite_std": 0.32534006237983704} {"timestamp_utc": "2026-04-12T00:00:23Z", "mode": "train", "global_step": 1331, "epoch": 0.05346025625577379, "loss": -0.0004, "grad_norm": 2.5885565280914307, "learning_rate": 5.96969696969697e-06, "num_tokens": 3003703.0, "completions/mean_length": 142.0, "completions/min_length": 135.0, "completions/max_length": 148.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 142.0, "completions/min_terminated_length": 135.0, "completions/max_terminated_length": 148.0, "rewards/meter/mean": 0.8050373792648315, "rewards/meter/std": 0.2827489972114563, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7777777910232544, "rewards/repeat_penalty/std": 0.08399210125207901, "rewards/total_composite/mean": 0.5267435908317566, "rewards/total_composite/std": 0.19810138642787933, "reward": 0.5267435908317566, "reward_std": 0.19810135662555695, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.032939184457063675, "sampling/sampling_logp_difference/max": 1.6217889785766602, "sampling/importance_sampling_ratio/min": 0.19754497706890106, "sampling/importance_sampling_ratio/mean": 0.9992785453796387, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1445506028831005, "clip_ratio/low_mean": 0.006243940908461809, "clip_ratio/low_min": 0.006243940908461809, "clip_ratio/high_mean": 0.028128837468102574, "clip_ratio/high_max": 0.028128837468102574, "clip_ratio/region_mean": 0.034372778376564384, "reward_total_mean": 0.5267435908317566, "reward_meter_mean": 0.8050373792648315, "reward_meter_std": 0.2827489972114563, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7777777910232544, "reward_repeat_penalty_std": 0.08399210125207901, "reward_total_composite_mean": 0.5267435908317566, "reward_total_composite_std": 0.19810138642787933} {"timestamp_utc": "2026-04-12T00:00:29Z", "mode": "train", "global_step": 1332, "epoch": 0.05350042173755874, "loss": 0.0143, "grad_norm": 5.604155540466309, "learning_rate": 5.966666666666667e-06, "num_tokens": 3006083.0, "completions/mean_length": 134.5, "completions/min_length": 129.0, "completions/max_length": 144.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 134.5, "completions/min_terminated_length": 129.0, "completions/max_terminated_length": 144.0, "rewards/meter/mean": 0.48692476749420166, "rewards/meter/std": 0.34781885147094727, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8214285373687744, "rewards/repeat_penalty/std": 0.10101525485515594, "rewards/total_composite/mean": 0.39111724495887756, "rewards/total_composite/std": 0.2647561728954315, "reward": 0.39111724495887756, "reward_std": 0.2647561728954315, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04181791469454765, "sampling/sampling_logp_difference/max": 0.9087643623352051, "sampling/importance_sampling_ratio/min": 0.403021901845932, "sampling/importance_sampling_ratio/mean": 1.007440209388733, "sampling/importance_sampling_ratio/max": 1.928771734237671, "entropy": 0.2525683268904686, "clip_ratio/low_mean": 0.015480266651138663, "clip_ratio/low_min": 0.015480266651138663, "clip_ratio/high_mean": 0.01978508895263076, "clip_ratio/high_max": 0.01978508895263076, "clip_ratio/region_mean": 0.03526535560376942, "reward_total_mean": 0.39111724495887756, "reward_meter_mean": 0.48692476749420166, "reward_meter_std": 0.34781885147094727, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8214285373687744, "reward_repeat_penalty_std": 0.10101525485515594, "reward_total_composite_mean": 0.39111724495887756, "reward_total_composite_std": 0.2647561728954315} {"timestamp_utc": "2026-04-12T00:00:34Z", "mode": "train", "global_step": 1333, "epoch": 0.0535405872193437, "loss": 0.0109, "grad_norm": 8.868410110473633, "learning_rate": 5.963636363636364e-06, "num_tokens": 3007848.0, "completions/mean_length": 59.625, "completions/min_length": 54.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.625, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.2418718934059143, "rewards/meter/std": 0.20975346863269806, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.22981885075569153, "rewards/total_composite/std": 0.2094106674194336, "reward": 0.22981885075569153, "reward_std": 0.2094106674194336, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06638707220554352, "sampling/sampling_logp_difference/max": 1.021674633026123, "sampling/importance_sampling_ratio/min": 0.3599916100502014, "sampling/importance_sampling_ratio/mean": 1.0142555236816406, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5658834353089333, "clip_ratio/low_mean": 0.041454336838796735, "clip_ratio/low_min": 0.041454336838796735, "clip_ratio/high_mean": 0.027125590480864048, "clip_ratio/high_max": 0.027125590480864048, "clip_ratio/region_mean": 0.06857992731966078, "reward_total_mean": 0.22981885075569153, "reward_meter_mean": 0.2418718934059143, "reward_meter_std": 0.20975346863269806, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.22981885075569153, "reward_total_composite_std": 0.2094106674194336} {"timestamp_utc": "2026-04-12T00:00:39Z", "mode": "train", "global_step": 1334, "epoch": 0.05358075270112865, "loss": -0.0233, "grad_norm": 4.963626384735107, "learning_rate": 5.960606060606061e-06, "num_tokens": 3010034.0, "completions/mean_length": 98.25, "completions/min_length": 83.0, "completions/max_length": 110.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.25, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 110.0, "rewards/meter/mean": 0.5051434636116028, "rewards/meter/std": 0.3091416358947754, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.1414213478565216, "rewards/total_composite/mean": 0.40488189458847046, "rewards/total_composite/std": 0.2789016366004944, "reward": 0.40488189458847046, "reward_std": 0.278901606798172, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04572216421365738, "sampling/sampling_logp_difference/max": 2.452214241027832, "sampling/importance_sampling_ratio/min": 0.08610272407531738, "sampling/importance_sampling_ratio/mean": 1.0055177211761475, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.29656792618334293, "clip_ratio/low_mean": 0.01672628615051508, "clip_ratio/low_min": 0.01672628615051508, "clip_ratio/high_mean": 0.020469399401918054, "clip_ratio/high_max": 0.020469399401918054, "clip_ratio/region_mean": 0.03719568555243313, "reward_total_mean": 0.40488189458847046, "reward_meter_mean": 0.5051434636116028, "reward_meter_std": 0.3091416358947754, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.1414213478565216, "reward_total_composite_mean": 0.40488189458847046, "reward_total_composite_std": 0.2789016366004944} {"timestamp_utc": "2026-04-12T00:00:45Z", "mode": "train", "global_step": 1335, "epoch": 0.053620918182913604, "loss": 0.02, "grad_norm": 4.286172866821289, "learning_rate": 5.9575757575757575e-06, "num_tokens": 3012715.0, "completions/mean_length": 141.125, "completions/min_length": 139.0, "completions/max_length": 149.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 141.125, "completions/min_terminated_length": 139.0, "completions/max_terminated_length": 149.0, "rewards/meter/mean": 0.891250491142273, "rewards/meter/std": 0.16874396800994873, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7142857313156128, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6366074681282043, "rewards/total_composite/std": 0.12053140252828598, "reward": 0.6366074681282043, "reward_std": 0.12053138017654419, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013281167484819889, "sampling/sampling_logp_difference/max": 1.791276454925537, "sampling/importance_sampling_ratio/min": 0.16674718260765076, "sampling/importance_sampling_ratio/mean": 1.0008882284164429, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0390528803691268, "clip_ratio/low_mean": 0.0008389261784031987, "clip_ratio/low_min": 0.0008389261784031987, "clip_ratio/high_mean": 0.007086224795784801, "clip_ratio/high_max": 0.007086224795784801, "clip_ratio/region_mean": 0.007925150974188, "reward_total_mean": 0.6366074681282043, "reward_meter_mean": 0.891250491142273, "reward_meter_std": 0.16874396800994873, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7142857313156128, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6366074681282043, "reward_total_composite_std": 0.12053140252828598} {"timestamp_utc": "2026-04-12T00:00:52Z", "mode": "train", "global_step": 1336, "epoch": 0.05366108366469856, "loss": -0.0184, "grad_norm": 3.4981236457824707, "learning_rate": 5.954545454545455e-06, "num_tokens": 3016037.0, "completions/mean_length": 219.25, "completions/min_length": 202.0, "completions/max_length": 235.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 219.25, "completions/min_terminated_length": 202.0, "completions/max_terminated_length": 235.0, "rewards/meter/mean": 0.9961979389190674, "rewards/meter/std": 0.0030242139473557472, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.0578637570142746, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8533653616905212, "rewards/repeat_penalty/std": 0.06368338316679001, "rewards/total_composite/mean": 0.7166441082954407, "rewards/total_composite/std": 0.06638932228088379, "reward": 0.7166441082954407, "reward_std": 0.06638932973146439, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03909594938158989, "sampling/sampling_logp_difference/max": 1.3188912868499756, "sampling/importance_sampling_ratio/min": 0.2674316465854645, "sampling/importance_sampling_ratio/mean": 1.0017598867416382, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.20540771167725325, "clip_ratio/low_mean": 0.014849004452116787, "clip_ratio/low_min": 0.014849004452116787, "clip_ratio/high_mean": 0.022171670105308294, "clip_ratio/high_max": 0.022171670105308294, "clip_ratio/region_mean": 0.03702067455742508, "reward_total_mean": 0.7166441082954407, "reward_meter_mean": 0.9961979389190674, "reward_meter_std": 0.0030242139473557472, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.0578637570142746, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8533653616905212, "reward_repeat_penalty_std": 0.06368338316679001, "reward_total_composite_mean": 0.7166441082954407, "reward_total_composite_std": 0.06638932228088379} {"timestamp_utc": "2026-04-12T00:01:00Z", "mode": "train", "global_step": 1337, "epoch": 0.05370124914648351, "loss": -0.0154, "grad_norm": 1.7854704856872559, "learning_rate": 5.951515151515151e-06, "num_tokens": 3020248.0, "completions/mean_length": 292.375, "completions/min_length": 277.0, "completions/max_length": 322.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 292.375, "completions/min_terminated_length": 277.0, "completions/max_terminated_length": 322.0, "rewards/meter/mean": 0.9976522922515869, "rewards/meter/std": 0.0010374977719038725, "rewards/count_adherence/mean": 0.5500000715255737, "rewards/count_adherence/std": 0.030860668048262596, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7024509906768799, "rewards/repeat_penalty/std": 0.09470956027507782, "rewards/total_composite/mean": 0.3849773406982422, "rewards/total_composite/std": 0.05099942535161972, "reward": 0.3849773406982422, "reward_std": 0.05099942535161972, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015756171196699142, "sampling/sampling_logp_difference/max": 1.7869529724121094, "sampling/importance_sampling_ratio/min": 0.16746968030929565, "sampling/importance_sampling_ratio/mean": 1.0004962682724, "sampling/importance_sampling_ratio/max": 1.787358045578003, "entropy": 0.10823496524244547, "clip_ratio/low_mean": 0.004836487962165847, "clip_ratio/low_min": 0.004836487962165847, "clip_ratio/high_mean": 0.007067535363603383, "clip_ratio/high_max": 0.007067535363603383, "clip_ratio/region_mean": 0.01190402332576923, "reward_total_mean": 0.3849773406982422, "reward_meter_mean": 0.9976522922515869, "reward_meter_std": 0.0010374977719038725, "reward_count_adherence_mean": 0.5500000715255737, "reward_count_adherence_std": 0.030860668048262596, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7024509906768799, "reward_repeat_penalty_std": 0.09470956027507782, "reward_total_composite_mean": 0.3849773406982422, "reward_total_composite_std": 0.05099942535161972} {"timestamp_utc": "2026-04-12T00:01:05Z", "mode": "train", "global_step": 1338, "epoch": 0.053741414628268466, "loss": 0.018, "grad_norm": 10.472454071044922, "learning_rate": 5.948484848484849e-06, "num_tokens": 3021936.0, "completions/mean_length": 50.0, "completions/min_length": 47.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.0, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.6975492238998413, "rewards/meter/std": 0.3380868434906006, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.6181282997131348, "rewards/total_composite/std": 0.2992437779903412, "reward": 0.6181282997131348, "reward_std": 0.2992437779903412, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0779750868678093, "sampling/sampling_logp_difference/max": 6.2996506690979, "sampling/importance_sampling_ratio/min": 0.0018369464669376612, "sampling/importance_sampling_ratio/mean": 0.9909482598304749, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.21370963286608458, "clip_ratio/low_mean": 0.03290023095905781, "clip_ratio/low_min": 0.03290023095905781, "clip_ratio/high_mean": 0.024901960510760546, "clip_ratio/high_max": 0.024901960510760546, "clip_ratio/region_mean": 0.057802191469818354, "reward_total_mean": 0.6181282997131348, "reward_meter_mean": 0.6975492238998413, "reward_meter_std": 0.3380868434906006, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_total_composite_mean": 0.6181282997131348, "reward_total_composite_std": 0.2992437779903412} {"timestamp_utc": "2026-04-12T00:01:10Z", "mode": "train", "global_step": 1339, "epoch": 0.05378158011005342, "loss": 0.0, "grad_norm": 2.900134325027466, "learning_rate": 5.9454545454545465e-06, "num_tokens": 3024310.0, "completions/mean_length": 115.75, "completions/min_length": 106.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.75, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.981162428855896, "rewards/meter/std": 0.015516383573412895, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7857142686843872, "rewards/repeat_penalty/std": 0.1322600096464157, "rewards/total_composite/mean": 0.7711630463600159, "rewards/total_composite/std": 0.1326371133327484, "reward": 0.7711630463600159, "reward_std": 0.1326371133327484, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030475087463855743, "sampling/sampling_logp_difference/max": 1.8554028272628784, "sampling/importance_sampling_ratio/min": 0.15638993680477142, "sampling/importance_sampling_ratio/mean": 0.9987893104553223, "sampling/importance_sampling_ratio/max": 1.8718475103378296, "entropy": 0.13537064753472805, "clip_ratio/low_mean": 0.01771387830376625, "clip_ratio/low_min": 0.01771387830376625, "clip_ratio/high_mean": 0.019089363981038332, "clip_ratio/high_max": 0.019089363981038332, "clip_ratio/region_mean": 0.03680324228480458, "reward_total_mean": 0.7711630463600159, "reward_meter_mean": 0.981162428855896, "reward_meter_std": 0.015516383573412895, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7857142686843872, "reward_repeat_penalty_std": 0.1322600096464157, "reward_total_composite_mean": 0.7711630463600159, "reward_total_composite_std": 0.1326371133327484} {"timestamp_utc": "2026-04-12T00:01:15Z", "mode": "train", "global_step": 1340, "epoch": 0.053821745591838374, "loss": 0.0012, "grad_norm": 2.7907607555389404, "learning_rate": 5.942424242424243e-06, "num_tokens": 3026369.0, "completions/mean_length": 90.375, "completions/min_length": 87.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.375, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.9822173118591309, "rewards/meter/std": 0.01175969373434782, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7857738733291626, "rewards/total_composite/std": 0.009407748468220234, "reward": 0.7857738733291626, "reward_std": 0.009407754056155682, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031274572014808655, "sampling/sampling_logp_difference/max": 1.289837121963501, "sampling/importance_sampling_ratio/min": 0.27531561255455017, "sampling/importance_sampling_ratio/mean": 1.0027801990509033, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.16755217872560024, "clip_ratio/low_mean": 0.01111179729923606, "clip_ratio/low_min": 0.01111179729923606, "clip_ratio/high_mean": 0.020700859487988055, "clip_ratio/high_max": 0.020700859487988055, "clip_ratio/region_mean": 0.031812656787224114, "reward_total_mean": 0.7857738733291626, "reward_meter_mean": 0.9822173118591309, "reward_meter_std": 0.01175969373434782, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7857738733291626, "reward_total_composite_std": 0.009407748468220234} {"timestamp_utc": "2026-04-12T00:01:20Z", "mode": "train", "global_step": 1341, "epoch": 0.05386191107362333, "loss": 0.0155, "grad_norm": 7.174100399017334, "learning_rate": 5.93939393939394e-06, "num_tokens": 3028190.0, "completions/mean_length": 60.625, "completions/min_length": 60.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.625, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.985583484172821, "rewards/meter/std": 0.008517028763890266, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.985583484172821, "rewards/total_composite/std": 0.008517028763890266, "reward": 0.985583484172821, "reward_std": 0.008517015725374222, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030380286276340485, "sampling/sampling_logp_difference/max": 1.27167809009552, "sampling/importance_sampling_ratio/min": 0.28036075830459595, "sampling/importance_sampling_ratio/mean": 1.005619764328003, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17691593896597624, "clip_ratio/low_mean": 0.010416667209938169, "clip_ratio/low_min": 0.010416667209938169, "clip_ratio/high_mean": 0.01639674766920507, "clip_ratio/high_max": 0.01639674766920507, "clip_ratio/region_mean": 0.026813414879143238, "reward_total_mean": 0.985583484172821, "reward_meter_mean": 0.985583484172821, "reward_meter_std": 0.008517028763890266, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.985583484172821, "reward_total_composite_std": 0.008517028763890266} {"timestamp_utc": "2026-04-12T00:01:25Z", "mode": "train", "global_step": 1342, "epoch": 0.05390207655540828, "loss": 0.0171, "grad_norm": 4.300682544708252, "learning_rate": 5.936363636363637e-06, "num_tokens": 3030079.0, "completions/mean_length": 69.125, "completions/min_length": 67.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.125, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.963999330997467, "rewards/meter/std": 0.005599203985184431, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.963999330997467, "rewards/total_composite/std": 0.005599203985184431, "reward": 0.963999330997467, "reward_std": 0.00559920072555542, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011351129971444607, "sampling/sampling_logp_difference/max": 0.8801417350769043, "sampling/importance_sampling_ratio/min": 0.4147241413593292, "sampling/importance_sampling_ratio/mean": 1.0003573894500732, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03910455713048577, "clip_ratio/low_mean": 0.003597308532334864, "clip_ratio/low_min": 0.003597308532334864, "clip_ratio/high_mean": 0.0018656715983524919, "clip_ratio/high_max": 0.0018656715983524919, "clip_ratio/region_mean": 0.005462980130687356, "reward_total_mean": 0.963999330997467, "reward_meter_mean": 0.963999330997467, "reward_meter_std": 0.005599203985184431, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.963999330997467, "reward_total_composite_std": 0.005599203985184431} {"timestamp_utc": "2026-04-12T00:01:30Z", "mode": "train", "global_step": 1343, "epoch": 0.053942242037193236, "loss": 0.029, "grad_norm": 3.7703700065612793, "learning_rate": 5.933333333333335e-06, "num_tokens": 3031857.0, "completions/mean_length": 66.25, "completions/min_length": 63.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.25, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.4224083423614502, "rewards/meter/std": 0.23921163380146027, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4224083423614502, "rewards/total_composite/std": 0.23921163380146027, "reward": 0.4224083423614502, "reward_std": 0.23921160399913788, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.059910353273153305, "sampling/sampling_logp_difference/max": 3.5546302795410156, "sampling/importance_sampling_ratio/min": 0.0285919439047575, "sampling/importance_sampling_ratio/mean": 0.9989436268806458, "sampling/importance_sampling_ratio/max": 1.9895342588424683, "entropy": 0.24532775953412056, "clip_ratio/low_mean": 0.01829817472025752, "clip_ratio/low_min": 0.01829817472025752, "clip_ratio/high_mean": 0.013258611783385277, "clip_ratio/high_max": 0.013258611783385277, "clip_ratio/region_mean": 0.0315567865036428, "reward_total_mean": 0.4224083423614502, "reward_meter_mean": 0.4224083423614502, "reward_meter_std": 0.23921163380146027, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.4224083423614502, "reward_total_composite_std": 0.23921163380146027} {"timestamp_utc": "2026-04-12T00:01:36Z", "mode": "train", "global_step": 1344, "epoch": 0.05398240751897819, "loss": 0.0095, "grad_norm": 6.5239176750183105, "learning_rate": 5.93030303030303e-06, "num_tokens": 3034322.0, "completions/mean_length": 140.125, "completions/min_length": 136.0, "completions/max_length": 143.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 140.125, "completions/min_terminated_length": 136.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.900975227355957, "rewards/meter/std": 0.25265270471572876, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.84794020652771, "rewards/total_composite/std": 0.24164395034313202, "reward": 0.84794020652771, "reward_std": 0.24164395034313202, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.054831597954034805, "sampling/sampling_logp_difference/max": 1.7345802783966064, "sampling/importance_sampling_ratio/min": 0.17647425830364227, "sampling/importance_sampling_ratio/mean": 1.0002692937850952, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.352537851780653, "clip_ratio/low_mean": 0.01140083302743733, "clip_ratio/low_min": 0.01140083302743733, "clip_ratio/high_mean": 0.033956656232476234, "clip_ratio/high_max": 0.033956656232476234, "clip_ratio/region_mean": 0.045357489259913564, "reward_total_mean": 0.84794020652771, "reward_meter_mean": 0.900975227355957, "reward_meter_std": 0.25265270471572876, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.84794020652771, "reward_total_composite_std": 0.24164395034313202} {"timestamp_utc": "2026-04-12T00:01:40Z", "mode": "train", "global_step": 1345, "epoch": 0.054022573000763144, "loss": -0.0074, "grad_norm": 3.739021062850952, "learning_rate": 5.927272727272728e-06, "num_tokens": 3036050.0, "completions/mean_length": 60.0, "completions/min_length": 59.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.0, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9513925313949585, "rewards/meter/std": 0.06354161351919174, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9513925313949585, "rewards/total_composite/std": 0.06354161351919174, "reward": 0.9513925313949585, "reward_std": 0.06354160606861115, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.040932025760412216, "sampling/sampling_logp_difference/max": 1.3692903518676758, "sampling/importance_sampling_ratio/min": 0.25428733229637146, "sampling/importance_sampling_ratio/mean": 0.9930896162986755, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.20175675675272942, "clip_ratio/low_mean": 0.0190677959471941, "clip_ratio/low_min": 0.0190677959471941, "clip_ratio/high_mean": 0.01858155010268092, "clip_ratio/high_max": 0.01858155010268092, "clip_ratio/region_mean": 0.03764934604987502, "reward_total_mean": 0.9513925313949585, "reward_meter_mean": 0.9513925313949585, "reward_meter_std": 0.06354161351919174, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9513925313949585, "reward_total_composite_std": 0.06354161351919174} {"timestamp_utc": "2026-04-12T00:01:46Z", "mode": "train", "global_step": 1346, "epoch": 0.0540627384825481, "loss": 0.0129, "grad_norm": 3.906574249267578, "learning_rate": 5.924242424242425e-06, "num_tokens": 3038243.0, "completions/mean_length": 132.125, "completions/min_length": 128.0, "completions/max_length": 134.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.125, "completions/min_terminated_length": 128.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.902446985244751, "rewards/meter/std": 0.14302004873752594, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.6613233089447021, "rewards/total_composite/std": 0.11659383773803711, "reward": 0.6613233089447021, "reward_std": 0.11659382283687592, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013300405815243721, "sampling/sampling_logp_difference/max": 1.1266558170318604, "sampling/importance_sampling_ratio/min": 0.3424758315086365, "sampling/importance_sampling_ratio/mean": 1.0032755136489868, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.04485098086297512, "clip_ratio/low_mean": 0.0009328357991762459, "clip_ratio/low_min": 0.0009328357991762459, "clip_ratio/high_mean": 0.011403630953282118, "clip_ratio/high_max": 0.011403630953282118, "clip_ratio/region_mean": 0.012336466752458364, "reward_total_mean": 0.6613233089447021, "reward_meter_mean": 0.902446985244751, "reward_meter_std": 0.14302004873752594, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.6613233089447021, "reward_total_composite_std": 0.11659383773803711} {"timestamp_utc": "2026-04-12T00:01:51Z", "mode": "train", "global_step": 1347, "epoch": 0.05410290396433305, "loss": 0.0011, "grad_norm": 3.224806070327759, "learning_rate": 5.921212121212122e-06, "num_tokens": 3040443.0, "completions/mean_length": 111.0, "completions/min_length": 108.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 111.0, "completions/min_terminated_length": 108.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.9691954851150513, "rewards/meter/std": 0.036640021950006485, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.9197216033935547, "rewards/total_composite/std": 0.08622133731842041, "reward": 0.9197216033935547, "reward_std": 0.08622132986783981, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03579000383615494, "sampling/sampling_logp_difference/max": 1.3639929294586182, "sampling/importance_sampling_ratio/min": 0.2556380033493042, "sampling/importance_sampling_ratio/mean": 1.011066198348999, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.27726688235998154, "clip_ratio/low_mean": 0.010106078116223216, "clip_ratio/low_min": 0.010106078116223216, "clip_ratio/high_mean": 0.024769189301878214, "clip_ratio/high_max": 0.024769189301878214, "clip_ratio/region_mean": 0.03487526741810143, "reward_total_mean": 0.9197216033935547, "reward_meter_mean": 0.9691954851150513, "reward_meter_std": 0.036640021950006485, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.9197216033935547, "reward_total_composite_std": 0.08622133731842041} {"timestamp_utc": "2026-04-12T00:01:56Z", "mode": "train", "global_step": 1348, "epoch": 0.054143069446118006, "loss": 0.011, "grad_norm": 4.513327121734619, "learning_rate": 5.9181818181818184e-06, "num_tokens": 3042433.0, "completions/mean_length": 71.75, "completions/min_length": 67.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.75, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.99162358045578, "rewards/meter/std": 0.011257669888436794, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.99162358045578, "rewards/total_composite/std": 0.011257669888436794, "reward": 0.99162358045578, "reward_std": 0.011257672682404518, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06130176782608032, "sampling/sampling_logp_difference/max": 1.3990116119384766, "sampling/importance_sampling_ratio/min": 0.24684081971645355, "sampling/importance_sampling_ratio/mean": 1.0111957788467407, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.404640544205904, "clip_ratio/low_mean": 0.006873097387142479, "clip_ratio/low_min": 0.006873097387142479, "clip_ratio/high_mean": 0.0280318089062348, "clip_ratio/high_max": 0.0280318089062348, "clip_ratio/region_mean": 0.03490490629337728, "reward_total_mean": 0.99162358045578, "reward_meter_mean": 0.99162358045578, "reward_meter_std": 0.011257669888436794, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.99162358045578, "reward_total_composite_std": 0.011257669888436794} {"timestamp_utc": "2026-04-12T00:02:01Z", "mode": "train", "global_step": 1349, "epoch": 0.05418323492790296, "loss": -0.0009, "grad_norm": 3.323253631591797, "learning_rate": 5.915151515151516e-06, "num_tokens": 3044432.0, "completions/mean_length": 89.875, "completions/min_length": 88.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.875, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.9421913623809814, "rewards/meter/std": 0.0664798840880394, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.753753125667572, "rewards/total_composite/std": 0.0531839095056057, "reward": 0.753753125667572, "reward_std": 0.0531838983297348, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014572631567716599, "sampling/sampling_logp_difference/max": 0.9959890842437744, "sampling/importance_sampling_ratio/min": 0.3693579435348511, "sampling/importance_sampling_ratio/mean": 0.9989533424377441, "sampling/importance_sampling_ratio/max": 1.4543837308883667, "entropy": 0.10254091210663319, "clip_ratio/low_mean": 0.004213483072817326, "clip_ratio/low_min": 0.004213483072817326, "clip_ratio/high_mean": 0.009848485118709505, "clip_ratio/high_max": 0.009848485118709505, "clip_ratio/region_mean": 0.01406196819152683, "reward_total_mean": 0.753753125667572, "reward_meter_mean": 0.9421913623809814, "reward_meter_std": 0.0664798840880394, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.753753125667572, "reward_total_composite_std": 0.0531839095056057} {"timestamp_utc": "2026-04-12T00:02:05Z", "mode": "train", "global_step": 1350, "epoch": 0.054223400409687914, "loss": 0.0046, "grad_norm": 4.666144847869873, "learning_rate": 5.912121212121212e-06, "num_tokens": 3046237.0, "completions/mean_length": 63.625, "completions/min_length": 61.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.625, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.5202029347419739, "rewards/meter/std": 0.32961180806159973, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5202029347419739, "rewards/total_composite/std": 0.32961180806159973, "reward": 0.5202029347419739, "reward_std": 0.32961177825927734, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03747888654470444, "sampling/sampling_logp_difference/max": 1.3192882537841797, "sampling/importance_sampling_ratio/min": 0.2673254907131195, "sampling/importance_sampling_ratio/mean": 0.9964258074760437, "sampling/importance_sampling_ratio/max": 1.43716299533844, "entropy": 0.18225187622010708, "clip_ratio/low_mean": 0.015552184893749654, "clip_ratio/low_min": 0.015552184893749654, "clip_ratio/high_mean": 0.007785359863191843, "clip_ratio/high_max": 0.007785359863191843, "clip_ratio/region_mean": 0.023337544756941497, "reward_total_mean": 0.5202029347419739, "reward_meter_mean": 0.5202029347419739, "reward_meter_std": 0.32961180806159973, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5202029347419739, "reward_total_composite_std": 0.32961180806159973} {"timestamp_utc": "2026-04-12T00:03:04Z", "mode": "eval", "global_step": 1350, "epoch": 0.054223400409687914, "eval_loss": NaN, "eval_runtime": 59.1029, "eval_samples_per_second": 1.76, "eval_steps_per_second": 0.22, "eval_num_tokens": 3046237.0, "eval_completions/mean_length": 172.09615384615384, "eval_completions/min_length": 61.46153846153846, "eval_completions/max_length": 307.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 172.09615384615384, "eval_completions/min_terminated_length": 61.46153846153846, "eval_completions/max_terminated_length": 307.0, "eval_rewards/meter/mean": 0.6744922651694372, "eval_rewards/meter/std": 0.41629832753768337, "eval_rewards/count_adherence/mean": 0.8681816137754, "eval_rewards/count_adherence/std": 0.14402351528406143, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/repeat_penalty/mean": 0.7968475085038406, "eval_rewards/repeat_penalty/std": 0.16538272224939787, "eval_rewards/total_composite/mean": 0.47231991015947783, "eval_rewards/total_composite/std": 0.3407691912009166, "eval_reward": 0.47231991015947783, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.016293376182707455, "eval_sampling/sampling_logp_difference/max": 1.005841695345365, "eval_sampling/importance_sampling_ratio/min": 0.37638011689369494, "eval_sampling/importance_sampling_ratio/mean": 1.0026835478269136, "eval_sampling/importance_sampling_ratio/max": 1.2998670798081617, "eval_entropy": 0.14777196657199126, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.47231991015947783, "eval_reward_meter_mean": 0.6744922651694372, "eval_reward_meter_std": 0.41629832753768337, "eval_reward_count_adherence_mean": 0.8681816137754, "eval_reward_count_adherence_std": 0.14402351528406143, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_repeat_penalty_mean": 0.7968475085038406, "eval_reward_repeat_penalty_std": 0.16538272224939787, "eval_reward_total_composite_mean": 0.47231991015947783, "eval_reward_total_composite_std": 0.3407691912009166} {"timestamp_utc": "2026-04-12T00:03:13Z", "mode": "train", "global_step": 1351, "epoch": 0.05426356589147287, "loss": 0.0176, "grad_norm": 3.602323532104492, "learning_rate": 5.90909090909091e-06, "num_tokens": 3048749.0, "completions/mean_length": 134.0, "completions/min_length": 128.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 134.0, "completions/min_terminated_length": 128.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.8950950503349304, "rewards/meter/std": 0.18451449275016785, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8928571939468384, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.7975612878799438, "rewards/total_composite/std": 0.16415363550186157, "reward": 0.7975612878799438, "reward_std": 0.16415365040302277, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.036450449377298355, "sampling/sampling_logp_difference/max": 1.9212646484375, "sampling/importance_sampling_ratio/min": 0.1464216709136963, "sampling/importance_sampling_ratio/mean": 1.0027098655700684, "sampling/importance_sampling_ratio/max": 1.8117905855178833, "entropy": 0.21928934194147587, "clip_ratio/low_mean": 0.008094369200989604, "clip_ratio/low_min": 0.008094369200989604, "clip_ratio/high_mean": 0.03014001634437591, "clip_ratio/high_max": 0.03014001634437591, "clip_ratio/region_mean": 0.03823438554536551, "reward_total_mean": 0.7975612878799438, "reward_meter_mean": 0.8950950503349304, "reward_meter_std": 0.18451449275016785, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8928571939468384, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.7975612878799438, "reward_total_composite_std": 0.16415363550186157} {"timestamp_utc": "2026-04-12T00:03:17Z", "mode": "train", "global_step": 1352, "epoch": 0.05430373137325782, "loss": 0.0112, "grad_norm": 3.5144999027252197, "learning_rate": 5.906060606060607e-06, "num_tokens": 3050215.0, "completions/mean_length": 32.25, "completions/min_length": 31.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.25, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9494646191596985, "rewards/meter/std": 0.08898033201694489, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9494646191596985, "rewards/total_composite/std": 0.08898033201694489, "reward": 0.9494646191596985, "reward_std": 0.08898033201694489, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03137016296386719, "sampling/sampling_logp_difference/max": 1.034781575202942, "sampling/importance_sampling_ratio/min": 0.35530397295951843, "sampling/importance_sampling_ratio/mean": 0.9998236298561096, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.14527534414082766, "clip_ratio/low_mean": 0.011482007801532745, "clip_ratio/low_min": 0.011482007801532745, "clip_ratio/high_mean": 0.011844757944345474, "clip_ratio/high_max": 0.011844757944345474, "clip_ratio/region_mean": 0.02332676574587822, "reward_total_mean": 0.9494646191596985, "reward_meter_mean": 0.9494646191596985, "reward_meter_std": 0.08898033201694489, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9494646191596985, "reward_total_composite_std": 0.08898033201694489} {"timestamp_utc": "2026-04-12T00:03:22Z", "mode": "train", "global_step": 1353, "epoch": 0.054343896855042775, "loss": 0.0173, "grad_norm": 7.210882186889648, "learning_rate": 5.903030303030304e-06, "num_tokens": 3052065.0, "completions/mean_length": 68.25, "completions/min_length": 64.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.25, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9697257876396179, "rewards/meter/std": 0.039108145982027054, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9697257876396179, "rewards/total_composite/std": 0.039108145982027054, "reward": 0.9697257876396179, "reward_std": 0.03910814970731735, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03900406137108803, "sampling/sampling_logp_difference/max": 1.7512474060058594, "sampling/importance_sampling_ratio/min": 0.173557311296463, "sampling/importance_sampling_ratio/mean": 1.003574252128601, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.19455979391932487, "clip_ratio/low_mean": 0.014525586506351829, "clip_ratio/low_min": 0.014525586506351829, "clip_ratio/high_mean": 0.031015241518616676, "clip_ratio/high_max": 0.031015241518616676, "clip_ratio/region_mean": 0.045540828024968505, "reward_total_mean": 0.9697257876396179, "reward_meter_mean": 0.9697257876396179, "reward_meter_std": 0.039108145982027054, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9697257876396179, "reward_total_composite_std": 0.039108145982027054} {"timestamp_utc": "2026-04-12T00:03:27Z", "mode": "train", "global_step": 1354, "epoch": 0.05438406233682773, "loss": 0.0013, "grad_norm": 2.7349421977996826, "learning_rate": 5.9e-06, "num_tokens": 3054440.0, "completions/mean_length": 128.875, "completions/min_length": 127.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.875, "completions/min_terminated_length": 127.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.9711472392082214, "rewards/meter/std": 0.022641118615865707, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7678571343421936, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.7450422048568726, "rewards/total_composite/std": 0.06550093740224838, "reward": 0.7450422048568726, "reward_std": 0.06550093740224838, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01926073245704174, "sampling/sampling_logp_difference/max": 1.9643745422363281, "sampling/importance_sampling_ratio/min": 0.14024357497692108, "sampling/importance_sampling_ratio/mean": 1.0016677379608154, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07222465332597494, "clip_ratio/low_mean": 0.004853119375184178, "clip_ratio/low_min": 0.004853119375184178, "clip_ratio/high_mean": 0.009631440974771976, "clip_ratio/high_max": 0.009631440974771976, "clip_ratio/region_mean": 0.014484560349956155, "reward_total_mean": 0.7450422048568726, "reward_meter_mean": 0.9711472392082214, "reward_meter_std": 0.022641118615865707, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7678571343421936, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.7450422048568726, "reward_total_composite_std": 0.06550093740224838} {"timestamp_utc": "2026-04-12T00:03:31Z", "mode": "train", "global_step": 1355, "epoch": 0.05442422781861268, "loss": 0.017, "grad_norm": 13.958351135253906, "learning_rate": 5.8969696969696975e-06, "num_tokens": 3055950.0, "completions/mean_length": 33.75, "completions/min_length": 32.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.75, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9982504844665527, "rewards/meter/std": 0.0002638747973833233, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9982504844665527, "rewards/total_composite/std": 0.0002638747973833233, "reward": 0.9982504844665527, "reward_std": 0.00026387875550426543, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0569852739572525, "sampling/sampling_logp_difference/max": 1.649427890777588, "sampling/importance_sampling_ratio/min": 0.19215981662273407, "sampling/importance_sampling_ratio/mean": 1.005803108215332, "sampling/importance_sampling_ratio/max": 1.8974801301956177, "entropy": 0.3209527414292097, "clip_ratio/low_mean": 0.011488970601931214, "clip_ratio/low_min": 0.011488970601931214, "clip_ratio/high_mean": 0.04795599193312228, "clip_ratio/high_max": 0.04795599193312228, "clip_ratio/region_mean": 0.05944496253505349, "reward_total_mean": 0.9982504844665527, "reward_meter_mean": 0.9982504844665527, "reward_meter_std": 0.0002638747973833233, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9982504844665527, "reward_total_composite_std": 0.0002638747973833233} {"timestamp_utc": "2026-04-12T00:03:35Z", "mode": "train", "global_step": 1356, "epoch": 0.05446439330039764, "loss": 0.0063, "grad_norm": 8.597617149353027, "learning_rate": 5.893939393939394e-06, "num_tokens": 3057379.0, "completions/mean_length": 31.625, "completions/min_length": 31.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.625, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9754250049591064, "rewards/meter/std": 0.010440711863338947, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9754250049591064, "rewards/total_composite/std": 0.010440711863338947, "reward": 0.9754250049591064, "reward_std": 0.010440702550113201, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02750094048678875, "sampling/sampling_logp_difference/max": 1.307356834411621, "sampling/importance_sampling_ratio/min": 0.2705341875553131, "sampling/importance_sampling_ratio/mean": 0.9994874596595764, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0775892292149365, "clip_ratio/low_mean": 0.023815523833036423, "clip_ratio/low_min": 0.023815523833036423, "clip_ratio/high_mean": 0.007938507944345474, "clip_ratio/high_max": 0.007938507944345474, "clip_ratio/region_mean": 0.0317540317773819, "reward_total_mean": 0.9754250049591064, "reward_meter_mean": 0.9754250049591064, "reward_meter_std": 0.010440711863338947, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9754250049591064, "reward_total_composite_std": 0.010440711863338947} {"timestamp_utc": "2026-04-12T00:03:40Z", "mode": "train", "global_step": 1357, "epoch": 0.05450455878218259, "loss": -0.0049, "grad_norm": 4.431914329528809, "learning_rate": 5.890909090909091e-06, "num_tokens": 3059262.0, "completions/mean_length": 61.375, "completions/min_length": 60.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.375, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.3558826744556427, "rewards/meter/std": 0.09776563197374344, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.3558826744556427, "rewards/total_composite/std": 0.09776563197374344, "reward": 0.3558826744556427, "reward_std": 0.09776563197374344, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04409278184175491, "sampling/sampling_logp_difference/max": 1.596221923828125, "sampling/importance_sampling_ratio/min": 0.20266073942184448, "sampling/importance_sampling_ratio/mean": 1.008901596069336, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2319873459637165, "clip_ratio/low_mean": 0.01653005462139845, "clip_ratio/low_min": 0.01653005462139845, "clip_ratio/high_mean": 0.028264103457331657, "clip_ratio/high_max": 0.028264103457331657, "clip_ratio/region_mean": 0.044794158078730106, "reward_total_mean": 0.3558826744556427, "reward_meter_mean": 0.3558826744556427, "reward_meter_std": 0.09776563197374344, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.3558826744556427, "reward_total_composite_std": 0.09776563197374344} {"timestamp_utc": "2026-04-12T00:03:44Z", "mode": "train", "global_step": 1358, "epoch": 0.054544724263967545, "loss": -0.003, "grad_norm": 6.204784393310547, "learning_rate": 5.887878787878788e-06, "num_tokens": 3061057.0, "completions/mean_length": 76.375, "completions/min_length": 74.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.375, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9842313528060913, "rewards/meter/std": 0.035786159336566925, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9842313528060913, "rewards/total_composite/std": 0.035786159336566925, "reward": 0.9842313528060913, "reward_std": 0.03578615561127663, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.042138196527957916, "sampling/sampling_logp_difference/max": 1.1789560317993164, "sampling/importance_sampling_ratio/min": 0.3075996935367584, "sampling/importance_sampling_ratio/mean": 1.0056933164596558, "sampling/importance_sampling_ratio/max": 1.8513555526733398, "entropy": 0.24777152389287949, "clip_ratio/low_mean": 0.0016891892300918698, "clip_ratio/low_min": 0.0016891892300918698, "clip_ratio/high_mean": 0.040843427646905184, "clip_ratio/high_max": 0.040843427646905184, "clip_ratio/region_mean": 0.042532616876997054, "reward_total_mean": 0.9842313528060913, "reward_meter_mean": 0.9842313528060913, "reward_meter_std": 0.035786159336566925, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9842313528060913, "reward_total_composite_std": 0.035786159336566925} {"timestamp_utc": "2026-04-12T00:03:49Z", "mode": "train", "global_step": 1359, "epoch": 0.0545848897457525, "loss": 0.0055, "grad_norm": 6.726247787475586, "learning_rate": 5.884848484848486e-06, "num_tokens": 3062858.0, "completions/mean_length": 72.125, "completions/min_length": 71.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.125, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.938990592956543, "rewards/meter/std": 0.15821680426597595, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.938990592956543, "rewards/total_composite/std": 0.15821680426597595, "reward": 0.938990592956543, "reward_std": 0.15821681916713715, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04102391377091408, "sampling/sampling_logp_difference/max": 1.6344432830810547, "sampling/importance_sampling_ratio/min": 0.19506093859672546, "sampling/importance_sampling_ratio/mean": 1.00753915309906, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31099611707031727, "clip_ratio/low_mean": 0.006849315017461777, "clip_ratio/low_min": 0.006849315017461777, "clip_ratio/high_mean": 0.026044346392154694, "clip_ratio/high_max": 0.026044346392154694, "clip_ratio/region_mean": 0.03289366140961647, "reward_total_mean": 0.938990592956543, "reward_meter_mean": 0.938990592956543, "reward_meter_std": 0.15821680426597595, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.938990592956543, "reward_total_composite_std": 0.15821680426597595} {"timestamp_utc": "2026-04-12T00:03:54Z", "mode": "train", "global_step": 1360, "epoch": 0.05462505522753745, "loss": 0.0188, "grad_norm": 6.566722869873047, "learning_rate": 5.881818181818182e-06, "num_tokens": 3064544.0, "completions/mean_length": 69.75, "completions/min_length": 65.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.75, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9943356513977051, "rewards/meter/std": 0.011215281672775745, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9943356513977051, "rewards/total_composite/std": 0.011215281672775745, "reward": 0.9943356513977051, "reward_std": 0.011215298436582088, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07216734439134598, "sampling/sampling_logp_difference/max": 1.3258538246154785, "sampling/importance_sampling_ratio/min": 0.2655761241912842, "sampling/importance_sampling_ratio/mean": 0.9980753064155579, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3794647231698036, "clip_ratio/low_mean": 0.0069444444961845875, "clip_ratio/low_min": 0.0069444444961845875, "clip_ratio/high_mean": 0.06222078762948513, "clip_ratio/high_max": 0.06222078762948513, "clip_ratio/region_mean": 0.06916523212566972, "reward_total_mean": 0.9943356513977051, "reward_meter_mean": 0.9943356513977051, "reward_meter_std": 0.011215281672775745, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9943356513977051, "reward_total_composite_std": 0.011215281672775745} {"timestamp_utc": "2026-04-12T00:03:58Z", "mode": "train", "global_step": 1361, "epoch": 0.05466522070932241, "loss": 0.0262, "grad_norm": 10.907339096069336, "learning_rate": 5.878787878787879e-06, "num_tokens": 3066089.0, "completions/mean_length": 39.125, "completions/min_length": 37.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.125, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.8477911949157715, "rewards/meter/std": 0.32613706588745117, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8477911949157715, "rewards/total_composite/std": 0.32613706588745117, "reward": 0.8477911949157715, "reward_std": 0.3261370360851288, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07346105575561523, "sampling/sampling_logp_difference/max": 1.893021583557129, "sampling/importance_sampling_ratio/min": 0.15061601996421814, "sampling/importance_sampling_ratio/mean": 1.0094305276870728, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40571761690080166, "clip_ratio/low_mean": 0.0062500000931322575, "clip_ratio/low_min": 0.0062500000931322575, "clip_ratio/high_mean": 0.051507277181372046, "clip_ratio/high_max": 0.051507277181372046, "clip_ratio/region_mean": 0.057757277274504304, "reward_total_mean": 0.8477911949157715, "reward_meter_mean": 0.8477911949157715, "reward_meter_std": 0.32613706588745117, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8477911949157715, "reward_total_composite_std": 0.32613706588745117} {"timestamp_utc": "2026-04-12T00:04:03Z", "mode": "train", "global_step": 1362, "epoch": 0.05470538619110736, "loss": 0.0211, "grad_norm": 5.575023651123047, "learning_rate": 5.875757575757576e-06, "num_tokens": 3068064.0, "completions/mean_length": 72.875, "completions/min_length": 67.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.875, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.8556216955184937, "rewards/meter/std": 0.27104073762893677, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.8141105771064758, "rewards/total_composite/std": 0.27185216546058655, "reward": 0.8141105771064758, "reward_std": 0.27185213565826416, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07884806394577026, "sampling/sampling_logp_difference/max": 1.8866146802902222, "sampling/importance_sampling_ratio/min": 0.15158410370349884, "sampling/importance_sampling_ratio/mean": 0.9950591921806335, "sampling/importance_sampling_ratio/max": 1.9909788370132446, "entropy": 0.3851870410144329, "clip_ratio/low_mean": 0.015579710714519024, "clip_ratio/low_min": 0.015579710714519024, "clip_ratio/high_mean": 0.06273440551012754, "clip_ratio/high_max": 0.06273440551012754, "clip_ratio/region_mean": 0.07831411622464657, "reward_total_mean": 0.8141105771064758, "reward_meter_mean": 0.8556216955184937, "reward_meter_std": 0.27104073762893677, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.8141105771064758, "reward_total_composite_std": 0.27185216546058655} {"timestamp_utc": "2026-04-12T00:04:07Z", "mode": "train", "global_step": 1363, "epoch": 0.054745551672892315, "loss": 0.0432, "grad_norm": 14.677414894104004, "learning_rate": 5.872727272727273e-06, "num_tokens": 3069688.0, "completions/mean_length": 32.0, "completions/min_length": 29.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.601068377494812, "rewards/meter/std": 0.3853451907634735, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.601068377494812, "rewards/total_composite/std": 0.3853451907634735, "reward": 0.601068377494812, "reward_std": 0.3853451907634735, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05407094210386276, "sampling/sampling_logp_difference/max": 1.9056625366210938, "sampling/importance_sampling_ratio/min": 0.14872407913208008, "sampling/importance_sampling_ratio/mean": 1.0056836605072021, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2922775577753782, "clip_ratio/low_mean": 0.010135134682059288, "clip_ratio/low_min": 0.010135134682059288, "clip_ratio/high_mean": 0.03211413393728435, "clip_ratio/high_max": 0.03211413393728435, "clip_ratio/region_mean": 0.04224926861934364, "reward_total_mean": 0.601068377494812, "reward_meter_mean": 0.601068377494812, "reward_meter_std": 0.3853451907634735, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.601068377494812, "reward_total_composite_std": 0.3853451907634735} {"timestamp_utc": "2026-04-12T00:04:12Z", "mode": "train", "global_step": 1364, "epoch": 0.05478571715467727, "loss": -0.0013, "grad_norm": 3.343250274658203, "learning_rate": 5.8696969696969694e-06, "num_tokens": 3071527.0, "completions/mean_length": 60.875, "completions/min_length": 60.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.875, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.987983763217926, "rewards/meter/std": 0.010048545897006989, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.987983763217926, "rewards/total_composite/std": 0.010048545897006989, "reward": 0.987983763217926, "reward_std": 0.010048558004200459, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020703302696347237, "sampling/sampling_logp_difference/max": 0.8887186050415039, "sampling/importance_sampling_ratio/min": 0.41118231415748596, "sampling/importance_sampling_ratio/mean": 0.9991443157196045, "sampling/importance_sampling_ratio/max": 1.870787501335144, "entropy": 0.10629169084131718, "clip_ratio/low_mean": 0.004132513655349612, "clip_ratio/low_min": 0.004132513655349612, "clip_ratio/high_mean": 0.010213951813057065, "clip_ratio/high_max": 0.010213951813057065, "clip_ratio/region_mean": 0.014346465468406677, "reward_total_mean": 0.987983763217926, "reward_meter_mean": 0.987983763217926, "reward_meter_std": 0.010048545897006989, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.987983763217926, "reward_total_composite_std": 0.010048545897006989} {"timestamp_utc": "2026-04-12T00:04:17Z", "mode": "train", "global_step": 1365, "epoch": 0.05482588263646222, "loss": 0.0227, "grad_norm": 5.388495445251465, "learning_rate": 5.8666666666666675e-06, "num_tokens": 3073505.0, "completions/mean_length": 91.25, "completions/min_length": 88.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.25, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.09923887252807617, "rewards/meter/std": 0.17276984453201294, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9249999523162842, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.09825273603200912, "rewards/total_composite/std": 0.17330721020698547, "reward": 0.09825273603200912, "reward_std": 0.17330722510814667, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05340426787734032, "sampling/sampling_logp_difference/max": 1.4949496984481812, "sampling/importance_sampling_ratio/min": 0.2242598831653595, "sampling/importance_sampling_ratio/mean": 1.003082036972046, "sampling/importance_sampling_ratio/max": 1.888500452041626, "entropy": 0.3218696843832731, "clip_ratio/low_mean": 0.032539316453039646, "clip_ratio/low_min": 0.032539316453039646, "clip_ratio/high_mean": 0.01694969367235899, "clip_ratio/high_max": 0.01694969367235899, "clip_ratio/region_mean": 0.049489010125398636, "reward_total_mean": 0.09825273603200912, "reward_meter_mean": 0.09923887252807617, "reward_meter_std": 0.17276984453201294, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9249999523162842, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.09825273603200912, "reward_total_composite_std": 0.17330721020698547} {"timestamp_utc": "2026-04-12T00:04:22Z", "mode": "train", "global_step": 1366, "epoch": 0.05486604811824718, "loss": 0.0137, "grad_norm": 8.658242225646973, "learning_rate": 5.863636363636364e-06, "num_tokens": 3075341.0, "completions/mean_length": 74.5, "completions/min_length": 69.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.5, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.9018104076385498, "rewards/meter/std": 0.15291984379291534, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9018104076385498, "rewards/total_composite/std": 0.15291984379291534, "reward": 0.9018104076385498, "reward_std": 0.15291984379291534, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08389697223901749, "sampling/sampling_logp_difference/max": 1.793445110321045, "sampling/importance_sampling_ratio/min": 0.1663859635591507, "sampling/importance_sampling_ratio/mean": 1.0000320672988892, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37352898344397545, "clip_ratio/low_mean": 0.019174665678292513, "clip_ratio/low_min": 0.019174665678292513, "clip_ratio/high_mean": 0.055123385740444064, "clip_ratio/high_max": 0.055123385740444064, "clip_ratio/region_mean": 0.07429805141873658, "reward_total_mean": 0.9018104076385498, "reward_meter_mean": 0.9018104076385498, "reward_meter_std": 0.15291984379291534, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9018104076385498, "reward_total_composite_std": 0.15291984379291534} {"timestamp_utc": "2026-04-12T00:04:26Z", "mode": "train", "global_step": 1367, "epoch": 0.05490621360003213, "loss": 0.0233, "grad_norm": 7.315425395965576, "learning_rate": 5.860606060606061e-06, "num_tokens": 3076962.0, "completions/mean_length": 62.625, "completions/min_length": 61.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.625, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9573122262954712, "rewards/meter/std": 0.04094774276018143, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9573122262954712, "rewards/total_composite/std": 0.04094774276018143, "reward": 0.9573122262954712, "reward_std": 0.04094773158431053, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03150878846645355, "sampling/sampling_logp_difference/max": 1.7324086427688599, "sampling/importance_sampling_ratio/min": 0.17685790359973907, "sampling/importance_sampling_ratio/mean": 0.994452714920044, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09856067411601543, "clip_ratio/low_mean": 0.013644688995555043, "clip_ratio/low_min": 0.013644688995555043, "clip_ratio/high_mean": 0.014278159011155367, "clip_ratio/high_max": 0.014278159011155367, "clip_ratio/region_mean": 0.02792284800671041, "reward_total_mean": 0.9573122262954712, "reward_meter_mean": 0.9573122262954712, "reward_meter_std": 0.04094774276018143, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9573122262954712, "reward_total_composite_std": 0.04094774276018143} {"timestamp_utc": "2026-04-12T00:04:33Z", "mode": "train", "global_step": 1368, "epoch": 0.054946379081817084, "loss": 0.0273, "grad_norm": 3.8533852100372314, "learning_rate": 5.8575757575757584e-06, "num_tokens": 3079903.0, "completions/mean_length": 186.625, "completions/min_length": 173.0, "completions/max_length": 218.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 186.625, "completions/min_terminated_length": 173.0, "completions/max_terminated_length": 218.0, "rewards/meter/mean": 0.7870528697967529, "rewards/meter/std": 0.30166095495224, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.07715168595314026, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9494949579238892, "rewards/repeat_penalty/std": 0.07303925603628159, "rewards/total_composite/mean": 0.644354522228241, "rewards/total_composite/std": 0.23647885024547577, "reward": 0.644354522228241, "reward_std": 0.23647886514663696, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07801329344511032, "sampling/sampling_logp_difference/max": 3.686089038848877, "sampling/importance_sampling_ratio/min": 0.025069857016205788, "sampling/importance_sampling_ratio/mean": 0.9927666783332825, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.35142070427536964, "clip_ratio/low_mean": 0.028853498864918947, "clip_ratio/low_min": 0.028853498864918947, "clip_ratio/high_mean": 0.03145654499530792, "clip_ratio/high_max": 0.03145654499530792, "clip_ratio/region_mean": 0.06031004386022687, "reward_total_mean": 0.644354522228241, "reward_meter_mean": 0.7870528697967529, "reward_meter_std": 0.30166095495224, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.07715168595314026, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9494949579238892, "reward_repeat_penalty_std": 0.07303925603628159, "reward_total_composite_mean": 0.644354522228241, "reward_total_composite_std": 0.23647885024547577} {"timestamp_utc": "2026-04-12T00:04:38Z", "mode": "train", "global_step": 1369, "epoch": 0.05498654456360204, "loss": -0.0105, "grad_norm": 6.314200401306152, "learning_rate": 5.854545454545455e-06, "num_tokens": 3082028.0, "completions/mean_length": 109.625, "completions/min_length": 104.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 109.625, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.09159435331821442, "rewards/meter/std": 0.1886763721704483, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.09159435331821442, "rewards/total_composite/std": 0.1886763721704483, "reward": 0.09159435331821442, "reward_std": 0.1886763721704483, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.056608088314533234, "sampling/sampling_logp_difference/max": 1.6070945262908936, "sampling/importance_sampling_ratio/min": 0.2004692256450653, "sampling/importance_sampling_ratio/mean": 1.008489966392517, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2641350254416466, "clip_ratio/low_mean": 0.04172989120706916, "clip_ratio/low_min": 0.04172989120706916, "clip_ratio/high_mean": 0.00881484942510724, "clip_ratio/high_max": 0.00881484942510724, "clip_ratio/region_mean": 0.0505447406321764, "reward_total_mean": 0.09159435331821442, "reward_meter_mean": 0.09159435331821442, "reward_meter_std": 0.1886763721704483, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.09159435331821442, "reward_total_composite_std": 0.1886763721704483} {"timestamp_utc": "2026-04-12T00:04:44Z", "mode": "train", "global_step": 1370, "epoch": 0.05502671004538699, "loss": 0.0008, "grad_norm": 2.687344551086426, "learning_rate": 5.851515151515152e-06, "num_tokens": 3084733.0, "completions/mean_length": 147.125, "completions/min_length": 129.0, "completions/max_length": 162.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 147.125, "completions/min_terminated_length": 129.0, "completions/max_terminated_length": 162.0, "rewards/meter/mean": 0.639522135257721, "rewards/meter/std": 0.19242770969867706, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7559523582458496, "rewards/repeat_penalty/std": 0.17674170434474945, "rewards/total_composite/mean": 0.4428256154060364, "rewards/total_composite/std": 0.15113258361816406, "reward": 0.4428256154060364, "reward_std": 0.15113258361816406, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030883800238370895, "sampling/sampling_logp_difference/max": 3.218033790588379, "sampling/importance_sampling_ratio/min": 0.040033698081970215, "sampling/importance_sampling_ratio/mean": 0.9980568289756775, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1116212410852313, "clip_ratio/low_mean": 0.013002454303205013, "clip_ratio/low_min": 0.013002454303205013, "clip_ratio/high_mean": 0.017049296759068966, "clip_ratio/high_max": 0.017049296759068966, "clip_ratio/region_mean": 0.03005175106227398, "reward_total_mean": 0.4428256154060364, "reward_meter_mean": 0.639522135257721, "reward_meter_std": 0.19242770969867706, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7559523582458496, "reward_repeat_penalty_std": 0.17674170434474945, "reward_total_composite_mean": 0.4428256154060364, "reward_total_composite_std": 0.15113258361816406} {"timestamp_utc": "2026-04-12T00:04:49Z", "mode": "train", "global_step": 1371, "epoch": 0.055066875527171946, "loss": 0.021, "grad_norm": 3.544004440307617, "learning_rate": 5.8484848484848485e-06, "num_tokens": 3087080.0, "completions/mean_length": 125.375, "completions/min_length": 119.0, "completions/max_length": 132.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 125.375, "completions/min_terminated_length": 119.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.7613728642463684, "rewards/meter/std": 0.35931819677352905, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.727212131023407, "rewards/total_composite/std": 0.34967657923698425, "reward": 0.727212131023407, "reward_std": 0.34967654943466187, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.034298431128263474, "sampling/sampling_logp_difference/max": 0.8778483867645264, "sampling/importance_sampling_ratio/min": 0.4156763255596161, "sampling/importance_sampling_ratio/mean": 1.009077787399292, "sampling/importance_sampling_ratio/max": 1.806413173675537, "entropy": 0.18958070501685143, "clip_ratio/low_mean": 0.010699023492634296, "clip_ratio/low_min": 0.010699023492634296, "clip_ratio/high_mean": 0.022983014350757003, "clip_ratio/high_max": 0.022983014350757003, "clip_ratio/region_mean": 0.0336820378433913, "reward_total_mean": 0.727212131023407, "reward_meter_mean": 0.7613728642463684, "reward_meter_std": 0.35931819677352905, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.727212131023407, "reward_total_composite_std": 0.34967657923698425} {"timestamp_utc": "2026-04-12T00:04:55Z", "mode": "train", "global_step": 1372, "epoch": 0.0551070410089569, "loss": 0.0079, "grad_norm": 5.1783366203308105, "learning_rate": 5.845454545454547e-06, "num_tokens": 3089717.0, "completions/mean_length": 153.625, "completions/min_length": 146.0, "completions/max_length": 159.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 153.625, "completions/min_terminated_length": 146.0, "completions/max_terminated_length": 159.0, "rewards/meter/mean": 0.05596386641263962, "rewards/meter/std": 0.034602247178554535, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.05596386641263962, "rewards/total_composite/std": 0.034602247178554535, "reward": 0.05596386641263962, "reward_std": 0.03460225090384483, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07258274406194687, "sampling/sampling_logp_difference/max": 5.408177852630615, "sampling/importance_sampling_ratio/min": 0.004479795694351196, "sampling/importance_sampling_ratio/mean": 0.9990571141242981, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.26759735494852066, "clip_ratio/low_mean": 0.009019391611218452, "clip_ratio/low_min": 0.009019391611218452, "clip_ratio/high_mean": 0.03647972596809268, "clip_ratio/high_max": 0.03647972596809268, "clip_ratio/region_mean": 0.04549911757931113, "reward_total_mean": 0.05596386641263962, "reward_meter_mean": 0.05596386641263962, "reward_meter_std": 0.034602247178554535, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.05596386641263962, "reward_total_composite_std": 0.034602247178554535} {"timestamp_utc": "2026-04-12T00:05:02Z", "mode": "train", "global_step": 1373, "epoch": 0.055147206490741854, "loss": 0.0156, "grad_norm": 2.5243654251098633, "learning_rate": 5.842424242424243e-06, "num_tokens": 3093370.0, "completions/mean_length": 265.625, "completions/min_length": 259.0, "completions/max_length": 272.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 265.625, "completions/min_terminated_length": 259.0, "completions/max_terminated_length": 272.0, "rewards/meter/mean": 0.9450531005859375, "rewards/meter/std": 0.09262127429246902, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7980769872665405, "rewards/repeat_penalty/std": 0.10019000619649887, "rewards/total_composite/mean": 0.6546376943588257, "rewards/total_composite/std": 0.05606427788734436, "reward": 0.6546376943588257, "reward_std": 0.05606428161263466, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.036455702036619186, "sampling/sampling_logp_difference/max": 3.1315479278564453, "sampling/importance_sampling_ratio/min": 0.043650176376104355, "sampling/importance_sampling_ratio/mean": 1.0016841888427734, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.22261576727032661, "clip_ratio/low_mean": 0.0139405254740268, "clip_ratio/low_min": 0.0139405254740268, "clip_ratio/high_mean": 0.01714563579298556, "clip_ratio/high_max": 0.01714563579298556, "clip_ratio/region_mean": 0.031086161267012358, "reward_total_mean": 0.6546376943588257, "reward_meter_mean": 0.9450531005859375, "reward_meter_std": 0.09262127429246902, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7980769872665405, "reward_repeat_penalty_std": 0.10019000619649887, "reward_total_composite_mean": 0.6546376943588257, "reward_total_composite_std": 0.05606427788734436} {"timestamp_utc": "2026-04-12T00:05:07Z", "mode": "train", "global_step": 1374, "epoch": 0.05518737197252681, "loss": 0.0443, "grad_norm": 5.411123752593994, "learning_rate": 5.83939393939394e-06, "num_tokens": 3095138.0, "completions/mean_length": 63.0, "completions/min_length": 61.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9784355759620667, "rewards/meter/std": 0.027008963748812675, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9784355759620667, "rewards/total_composite/std": 0.027008963748812675, "reward": 0.9784355759620667, "reward_std": 0.027008960023522377, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03236331045627594, "sampling/sampling_logp_difference/max": 1.732818365097046, "sampling/importance_sampling_ratio/min": 0.17678546905517578, "sampling/importance_sampling_ratio/mean": 1.0000766515731812, "sampling/importance_sampling_ratio/max": 1.9240130186080933, "entropy": 0.1000348855741322, "clip_ratio/low_mean": 0.011315496172755957, "clip_ratio/low_min": 0.011315496172755957, "clip_ratio/high_mean": 0.020197488833218813, "clip_ratio/high_max": 0.020197488833218813, "clip_ratio/region_mean": 0.03151298500597477, "reward_total_mean": 0.9784355759620667, "reward_meter_mean": 0.9784355759620667, "reward_meter_std": 0.027008963748812675, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9784355759620667, "reward_total_composite_std": 0.027008963748812675} {"timestamp_utc": "2026-04-12T00:05:11Z", "mode": "train", "global_step": 1375, "epoch": 0.05522753745431176, "loss": 0.0116, "grad_norm": 10.806543350219727, "learning_rate": 5.836363636363637e-06, "num_tokens": 3096588.0, "completions/mean_length": 40.25, "completions/min_length": 36.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.25, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9826058149337769, "rewards/meter/std": 0.01861639879643917, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9826058149337769, "rewards/total_composite/std": 0.01861639879643917, "reward": 0.9826058149337769, "reward_std": 0.018616393208503723, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06725956499576569, "sampling/sampling_logp_difference/max": 1.3660998344421387, "sampling/importance_sampling_ratio/min": 0.25509995222091675, "sampling/importance_sampling_ratio/mean": 1.0129060745239258, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47422025352716446, "clip_ratio/low_mean": 0.00937500037252903, "clip_ratio/low_min": 0.00937500037252903, "clip_ratio/high_mean": 0.04311362677253783, "clip_ratio/high_max": 0.04311362677253783, "clip_ratio/region_mean": 0.05248862714506686, "reward_total_mean": 0.9826058149337769, "reward_meter_mean": 0.9826058149337769, "reward_meter_std": 0.01861639879643917, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9826058149337769, "reward_total_composite_std": 0.01861639879643917} {"timestamp_utc": "2026-04-12T00:05:18Z", "mode": "train", "global_step": 1376, "epoch": 0.055267702936096716, "loss": 0.026, "grad_norm": 4.0455732345581055, "learning_rate": 5.833333333333334e-06, "num_tokens": 3100217.0, "completions/mean_length": 245.625, "completions/min_length": 241.0, "completions/max_length": 256.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 245.625, "completions/min_terminated_length": 241.0, "completions/max_terminated_length": 256.0, "rewards/meter/mean": 0.735993504524231, "rewards/meter/std": 0.3604574203491211, "rewards/count_adherence/mean": 0.7272727489471436, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.10079052299261093, "rewards/total_composite/mean": 0.4196595549583435, "rewards/total_composite/std": 0.19926168024539948, "reward": 0.4196595549583435, "reward_std": 0.19926169514656067, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06053018942475319, "sampling/sampling_logp_difference/max": 8.299843788146973, "sampling/importance_sampling_ratio/min": 0.00024855564697645605, "sampling/importance_sampling_ratio/mean": 0.9991115927696228, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2056045476347208, "clip_ratio/low_mean": 0.021116425283253193, "clip_ratio/low_min": 0.021116425283253193, "clip_ratio/high_mean": 0.019607282243669033, "clip_ratio/high_max": 0.019607282243669033, "clip_ratio/region_mean": 0.040723707526922226, "reward_total_mean": 0.4196595549583435, "reward_meter_mean": 0.735993504524231, "reward_meter_std": 0.3604574203491211, "reward_count_adherence_mean": 0.7272727489471436, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.10079052299261093, "reward_total_composite_mean": 0.4196595549583435, "reward_total_composite_std": 0.19926168024539948} {"timestamp_utc": "2026-04-12T00:05:23Z", "mode": "train", "global_step": 1377, "epoch": 0.05530786841788167, "loss": 0.0091, "grad_norm": 5.782622814178467, "learning_rate": 5.83030303030303e-06, "num_tokens": 3102173.0, "completions/mean_length": 79.5, "completions/min_length": 76.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.5, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.9965003728866577, "rewards/meter/std": 0.0018758656224235892, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9965003728866577, "rewards/total_composite/std": 0.0018758656224235892, "reward": 0.9965003728866577, "reward_std": 0.0018758544465526938, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.057900648564100266, "sampling/sampling_logp_difference/max": 2.173585891723633, "sampling/importance_sampling_ratio/min": 0.11376892775297165, "sampling/importance_sampling_ratio/mean": 1.003879189491272, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3501878045499325, "clip_ratio/low_mean": 0.02467968501150608, "clip_ratio/low_min": 0.02467968501150608, "clip_ratio/high_mean": 0.028373429435305297, "clip_ratio/high_max": 0.028373429435305297, "clip_ratio/region_mean": 0.05305311444681138, "reward_total_mean": 0.9965003728866577, "reward_meter_mean": 0.9965003728866577, "reward_meter_std": 0.0018758656224235892, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9965003728866577, "reward_total_composite_std": 0.0018758656224235892} {"timestamp_utc": "2026-04-12T00:05:29Z", "mode": "train", "global_step": 1378, "epoch": 0.055348033899666624, "loss": -0.0018, "grad_norm": 6.239497184753418, "learning_rate": 5.8272727272727285e-06, "num_tokens": 3104993.0, "completions/mean_length": 172.5, "completions/min_length": 142.0, "completions/max_length": 189.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 172.5, "completions/min_terminated_length": 142.0, "completions/max_terminated_length": 189.0, "rewards/meter/mean": 0.9290133714675903, "rewards/meter/std": 0.17080947756767273, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333134651184, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.8381317853927612, "rewards/total_composite/std": 0.152348130941391, "reward": 0.8381317853927612, "reward_std": 0.1523481160402298, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04662080481648445, "sampling/sampling_logp_difference/max": 2.5756921768188477, "sampling/importance_sampling_ratio/min": 0.07610113173723221, "sampling/importance_sampling_ratio/mean": 1.0066179037094116, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2811085060238838, "clip_ratio/low_mean": 0.010481631616130471, "clip_ratio/low_min": 0.010481631616130471, "clip_ratio/high_mean": 0.022769354982301593, "clip_ratio/high_max": 0.022769354982301593, "clip_ratio/region_mean": 0.033250986598432064, "reward_total_mean": 0.8381317853927612, "reward_meter_mean": 0.9290133714675903, "reward_meter_std": 0.17080947756767273, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333134651184, "reward_repeat_penalty_std": 0.05750546231865883, "reward_total_composite_mean": 0.8381317853927612, "reward_total_composite_std": 0.152348130941391} {"timestamp_utc": "2026-04-12T00:05:35Z", "mode": "train", "global_step": 1379, "epoch": 0.05538819938145158, "loss": -0.0035, "grad_norm": 2.3529670238494873, "learning_rate": 5.824242424242425e-06, "num_tokens": 3108711.0, "completions/mean_length": 241.75, "completions/min_length": 226.0, "completions/max_length": 248.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 241.75, "completions/min_terminated_length": 226.0, "completions/max_terminated_length": 248.0, "rewards/meter/mean": 0.9438363313674927, "rewards/meter/std": 0.13904310762882233, "rewards/count_adherence/mean": 0.7272727489471436, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7250000238418579, "rewards/repeat_penalty/std": 0.115125872194767, "rewards/total_composite/mean": 0.4973941147327423, "rewards/total_composite/std": 0.1098327562212944, "reward": 0.4973941147327423, "reward_std": 0.1098327562212944, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019148869439959526, "sampling/sampling_logp_difference/max": 1.7155005931854248, "sampling/importance_sampling_ratio/min": 0.17987366020679474, "sampling/importance_sampling_ratio/mean": 1.0025800466537476, "sampling/importance_sampling_ratio/max": 1.9625972509384155, "entropy": 0.09616232290863991, "clip_ratio/low_mean": 0.007781740743666887, "clip_ratio/low_min": 0.007781740743666887, "clip_ratio/high_mean": 0.006693349685519934, "clip_ratio/high_max": 0.006693349685519934, "clip_ratio/region_mean": 0.014475090429186821, "reward_total_mean": 0.4973941147327423, "reward_meter_mean": 0.9438363313674927, "reward_meter_std": 0.13904310762882233, "reward_count_adherence_mean": 0.7272727489471436, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7250000238418579, "reward_repeat_penalty_std": 0.115125872194767, "reward_total_composite_mean": 0.4973941147327423, "reward_total_composite_std": 0.1098327562212944} {"timestamp_utc": "2026-04-12T00:05:40Z", "mode": "train", "global_step": 1380, "epoch": 0.05542836486323653, "loss": 0.0058, "grad_norm": 3.860546112060547, "learning_rate": 5.821212121212122e-06, "num_tokens": 3110495.0, "completions/mean_length": 69.0, "completions/min_length": 68.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.0, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.014178031124174595, "rewards/meter/std": 0.004543017130345106, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.014178031124174595, "rewards/total_composite/std": 0.004543017130345106, "reward": 0.014178031124174595, "reward_std": 0.004543016664683819, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02862457185983658, "sampling/sampling_logp_difference/max": 2.9717161655426025, "sampling/importance_sampling_ratio/min": 0.051215339452028275, "sampling/importance_sampling_ratio/mean": 0.9993897676467896, "sampling/importance_sampling_ratio/max": 1.7626913785934448, "entropy": 0.1250211326405406, "clip_ratio/low_mean": 0.01270932296756655, "clip_ratio/low_min": 0.01270932296756655, "clip_ratio/high_mean": 0.008928571827709675, "clip_ratio/high_max": 0.008928571827709675, "clip_ratio/region_mean": 0.021637894795276225, "reward_total_mean": 0.014178031124174595, "reward_meter_mean": 0.014178031124174595, "reward_meter_std": 0.004543017130345106, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.014178031124174595, "reward_total_composite_std": 0.004543017130345106} {"timestamp_utc": "2026-04-12T00:05:45Z", "mode": "train", "global_step": 1381, "epoch": 0.055468530345021486, "loss": 0.0024, "grad_norm": 1.1249967813491821, "learning_rate": 5.8181818181818185e-06, "num_tokens": 3112713.0, "completions/mean_length": 115.25, "completions/min_length": 115.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.25, "completions/min_terminated_length": 115.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.9941756725311279, "rewards/meter/std": 0.00028240663232281804, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.3214285969734192, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.319552481174469, "rewards/total_composite/std": 0.0657178983092308, "reward": 0.319552481174469, "reward_std": 0.06571788340806961, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009904337115585804, "sampling/sampling_logp_difference/max": 1.0573194026947021, "sampling/importance_sampling_ratio/min": 0.3473857343196869, "sampling/importance_sampling_ratio/mean": 0.9989582896232605, "sampling/importance_sampling_ratio/max": 1.2996612787246704, "entropy": 0.039332801941782236, "clip_ratio/low_mean": 0.003251499147154391, "clip_ratio/low_min": 0.003251499147154391, "clip_ratio/high_mean": 0.003260869416408241, "clip_ratio/high_max": 0.003260869416408241, "clip_ratio/region_mean": 0.006512368563562632, "reward_total_mean": 0.319552481174469, "reward_meter_mean": 0.9941756725311279, "reward_meter_std": 0.00028240663232281804, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.3214285969734192, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.319552481174469, "reward_total_composite_std": 0.0657178983092308} {"timestamp_utc": "2026-04-12T00:05:50Z", "mode": "train", "global_step": 1382, "epoch": 0.05550869582680644, "loss": 0.0069, "grad_norm": 4.528812408447266, "learning_rate": 5.815151515151516e-06, "num_tokens": 3115264.0, "completions/mean_length": 151.875, "completions/min_length": 149.0, "completions/max_length": 155.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 151.875, "completions/min_terminated_length": 149.0, "completions/max_terminated_length": 155.0, "rewards/meter/mean": 0.9926921725273132, "rewards/meter/std": 0.008148393593728542, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7857142686843872, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.7798627614974976, "rewards/total_composite/std": 0.07480987906455994, "reward": 0.7798627614974976, "reward_std": 0.07480985671281815, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030164044350385666, "sampling/sampling_logp_difference/max": 1.2373886108398438, "sampling/importance_sampling_ratio/min": 0.29014089703559875, "sampling/importance_sampling_ratio/mean": 1.0049320459365845, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1818629987537861, "clip_ratio/low_mean": 0.0058121298789046705, "clip_ratio/low_min": 0.0058121298789046705, "clip_ratio/high_mean": 0.017227754462510347, "clip_ratio/high_max": 0.017227754462510347, "clip_ratio/region_mean": 0.023039884341415018, "reward_total_mean": 0.7798627614974976, "reward_meter_mean": 0.9926921725273132, "reward_meter_std": 0.008148393593728542, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7857142686843872, "reward_repeat_penalty_std": 0.07636035233736038, "reward_total_composite_mean": 0.7798627614974976, "reward_total_composite_std": 0.07480987906455994} {"timestamp_utc": "2026-04-12T00:05:55Z", "mode": "train", "global_step": 1383, "epoch": 0.055548861308591393, "loss": 0.0148, "grad_norm": 3.192974805831909, "learning_rate": 5.812121212121212e-06, "num_tokens": 3117462.0, "completions/mean_length": 103.75, "completions/min_length": 102.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.75, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.9773154258728027, "rewards/meter/std": 0.0542588047683239, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9773154258728027, "rewards/total_composite/std": 0.0542588047683239, "reward": 0.9773154258728027, "reward_std": 0.0542587973177433, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03143962472677231, "sampling/sampling_logp_difference/max": 1.3268545866012573, "sampling/importance_sampling_ratio/min": 0.331994891166687, "sampling/importance_sampling_ratio/mean": 1.0025712251663208, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.18817367777228355, "clip_ratio/low_mean": 0.001168224262073636, "clip_ratio/low_min": 0.001168224262073636, "clip_ratio/high_mean": 0.02543990476988256, "clip_ratio/high_max": 0.02543990476988256, "clip_ratio/region_mean": 0.026608129031956196, "reward_total_mean": 0.9773154258728027, "reward_meter_mean": 0.9773154258728027, "reward_meter_std": 0.0542588047683239, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9773154258728027, "reward_total_composite_std": 0.0542588047683239} {"timestamp_utc": "2026-04-12T00:06:00Z", "mode": "train", "global_step": 1384, "epoch": 0.05558902679037635, "loss": 0.0122, "grad_norm": 7.283441543579102, "learning_rate": 5.8090909090909095e-06, "num_tokens": 3119369.0, "completions/mean_length": 69.375, "completions/min_length": 66.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.375, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.99678635597229, "rewards/meter/std": 0.001158631988801062, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.99678635597229, "rewards/total_composite/std": 0.001158631988801062, "reward": 0.99678635597229, "reward_std": 0.001158619881607592, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.042962342500686646, "sampling/sampling_logp_difference/max": 4.707744598388672, "sampling/importance_sampling_ratio/min": 0.009025109931826591, "sampling/importance_sampling_ratio/mean": 1.0015077590942383, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.13980651926249266, "clip_ratio/low_mean": 0.014534936985000968, "clip_ratio/low_min": 0.014534936985000968, "clip_ratio/high_mean": 0.01638579391874373, "clip_ratio/high_max": 0.01638579391874373, "clip_ratio/region_mean": 0.030920730903744698, "reward_total_mean": 0.99678635597229, "reward_meter_mean": 0.99678635597229, "reward_meter_std": 0.001158631988801062, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.99678635597229, "reward_total_composite_std": 0.001158631988801062} {"timestamp_utc": "2026-04-12T00:06:04Z", "mode": "train", "global_step": 1385, "epoch": 0.0556291922721613, "loss": 0.0196, "grad_norm": 8.730010032653809, "learning_rate": 5.806060606060606e-06, "num_tokens": 3120803.0, "completions/mean_length": 32.25, "completions/min_length": 31.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.25, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9718137979507446, "rewards/meter/std": 0.05166900157928467, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9718137979507446, "rewards/total_composite/std": 0.05166900157928467, "reward": 0.9718137979507446, "reward_std": 0.051669009029865265, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01950211450457573, "sampling/sampling_logp_difference/max": 0.6880896091461182, "sampling/importance_sampling_ratio/min": 0.5025351643562317, "sampling/importance_sampling_ratio/mean": 1.0018343925476074, "sampling/importance_sampling_ratio/max": 1.604980230331421, "entropy": 0.1377080613747239, "clip_ratio/low_mean": 0.007464349502697587, "clip_ratio/low_min": 0.007464349502697587, "clip_ratio/high_mean": 0.00390625, "clip_ratio/high_max": 0.00390625, "clip_ratio/region_mean": 0.011370599502697587, "reward_total_mean": 0.9718137979507446, "reward_meter_mean": 0.9718137979507446, "reward_meter_std": 0.05166900157928467, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9718137979507446, "reward_total_composite_std": 0.05166900157928467} {"timestamp_utc": "2026-04-12T00:06:09Z", "mode": "train", "global_step": 1386, "epoch": 0.055669357753946255, "loss": -0.0023, "grad_norm": 5.677158832550049, "learning_rate": 5.803030303030304e-06, "num_tokens": 3122583.0, "completions/mean_length": 58.5, "completions/min_length": 55.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.5, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.307847261428833, "rewards/meter/std": 0.2652658224105835, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.307847261428833, "rewards/total_composite/std": 0.2652658224105835, "reward": 0.307847261428833, "reward_std": 0.2652658224105835, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04078377038240433, "sampling/sampling_logp_difference/max": 1.0301332473754883, "sampling/importance_sampling_ratio/min": 0.35695940256118774, "sampling/importance_sampling_ratio/mean": 1.0097620487213135, "sampling/importance_sampling_ratio/max": 1.7868449687957764, "entropy": 0.29168289341032505, "clip_ratio/low_mean": 0.017045454820618033, "clip_ratio/low_min": 0.017045454820618033, "clip_ratio/high_mean": 0.010490705259144306, "clip_ratio/high_max": 0.010490705259144306, "clip_ratio/region_mean": 0.02753616007976234, "reward_total_mean": 0.307847261428833, "reward_meter_mean": 0.307847261428833, "reward_meter_std": 0.2652658224105835, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.307847261428833, "reward_total_composite_std": 0.2652658224105835} {"timestamp_utc": "2026-04-12T00:06:16Z", "mode": "train", "global_step": 1387, "epoch": 0.05570952323573121, "loss": 0.0117, "grad_norm": 3.5950727462768555, "learning_rate": 5.8e-06, "num_tokens": 3125864.0, "completions/mean_length": 223.125, "completions/min_length": 206.0, "completions/max_length": 235.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 223.125, "completions/min_terminated_length": 206.0, "completions/max_terminated_length": 235.0, "rewards/meter/mean": 0.9973956346511841, "rewards/meter/std": 0.0017723083728924394, "rewards/count_adherence/mean": 0.8055555820465088, "rewards/count_adherence/std": 0.05143444985151291, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6742216348648071, "rewards/repeat_penalty/std": 0.1199769675731659, "rewards/total_composite/mean": 0.5415959358215332, "rewards/total_composite/std": 0.10306404531002045, "reward": 0.5415959358215332, "reward_std": 0.10306404531002045, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.058670658618211746, "sampling/sampling_logp_difference/max": 17.84847640991211, "sampling/importance_sampling_ratio/min": 1.7721692557870483e-08, "sampling/importance_sampling_ratio/mean": 1.0015687942504883, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.16380778048187494, "clip_ratio/low_mean": 0.011592080816626549, "clip_ratio/low_min": 0.011592080816626549, "clip_ratio/high_mean": 0.01465759938582778, "clip_ratio/high_max": 0.01465759938582778, "clip_ratio/region_mean": 0.02624968020245433, "reward_total_mean": 0.5415959358215332, "reward_meter_mean": 0.9973956346511841, "reward_meter_std": 0.0017723083728924394, "reward_count_adherence_mean": 0.8055555820465088, "reward_count_adherence_std": 0.05143444985151291, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6742216348648071, "reward_repeat_penalty_std": 0.1199769675731659, "reward_total_composite_mean": 0.5415959358215332, "reward_total_composite_std": 0.10306404531002045} {"timestamp_utc": "2026-04-12T00:06:20Z", "mode": "train", "global_step": 1388, "epoch": 0.05574968871751616, "loss": 0.0122, "grad_norm": 3.6467227935791016, "learning_rate": 5.796969696969698e-06, "num_tokens": 3127767.0, "completions/mean_length": 66.875, "completions/min_length": 66.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.875, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9714630842208862, "rewards/meter/std": 0.009149931371212006, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9714630842208862, "rewards/total_composite/std": 0.009149931371212006, "reward": 0.9714630842208862, "reward_std": 0.009149930439889431, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01836930215358734, "sampling/sampling_logp_difference/max": 0.6224894523620605, "sampling/importance_sampling_ratio/min": 0.5366069078445435, "sampling/importance_sampling_ratio/mean": 1.0066375732421875, "sampling/importance_sampling_ratio/max": 1.4535624980926514, "entropy": 0.12053523678332567, "clip_ratio/low_mean": 0.007462686393409967, "clip_ratio/low_min": 0.007462686393409967, "clip_ratio/high_mean": 0.009384893695823848, "clip_ratio/high_max": 0.009384893695823848, "clip_ratio/region_mean": 0.016847580089233816, "reward_total_mean": 0.9714630842208862, "reward_meter_mean": 0.9714630842208862, "reward_meter_std": 0.009149931371212006, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9714630842208862, "reward_total_composite_std": 0.009149931371212006} {"timestamp_utc": "2026-04-12T00:06:28Z", "mode": "train", "global_step": 1389, "epoch": 0.055789854199301124, "loss": -0.0159, "grad_norm": 2.405141592025757, "learning_rate": 5.793939393939394e-06, "num_tokens": 3131978.0, "completions/mean_length": 276.375, "completions/min_length": 259.0, "completions/max_length": 295.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 276.375, "completions/min_terminated_length": 259.0, "completions/max_terminated_length": 295.0, "rewards/meter/mean": 0.9949830770492554, "rewards/meter/std": 0.0029864103998988867, "rewards/count_adherence/mean": 0.7272727489471436, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7510416507720947, "rewards/repeat_penalty/std": 0.10675939172506332, "rewards/total_composite/mean": 0.5435611009597778, "rewards/total_composite/std": 0.07788535207509995, "reward": 0.5435611009597778, "reward_std": 0.07788533717393875, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028575100004673004, "sampling/sampling_logp_difference/max": 1.6693830490112305, "sampling/importance_sampling_ratio/min": 0.1883632391691208, "sampling/importance_sampling_ratio/mean": 1.002191185951233, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1869909018278122, "clip_ratio/low_mean": 0.010067827010061592, "clip_ratio/low_min": 0.010067827010061592, "clip_ratio/high_mean": 0.009672214509919286, "clip_ratio/high_max": 0.009672214509919286, "clip_ratio/region_mean": 0.019740041519980878, "reward_total_mean": 0.5435611009597778, "reward_meter_mean": 0.9949830770492554, "reward_meter_std": 0.0029864103998988867, "reward_count_adherence_mean": 0.7272727489471436, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7510416507720947, "reward_repeat_penalty_std": 0.10675939172506332, "reward_total_composite_mean": 0.5435611009597778, "reward_total_composite_std": 0.07788535207509995} {"timestamp_utc": "2026-04-12T00:06:33Z", "mode": "train", "global_step": 1390, "epoch": 0.05583001968108608, "loss": 0.0238, "grad_norm": 6.391792297363281, "learning_rate": 5.790909090909091e-06, "num_tokens": 3133718.0, "completions/mean_length": 67.5, "completions/min_length": 64.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.5, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.11840285360813141, "rewards/meter/std": 0.15882401168346405, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.11840285360813141, "rewards/total_composite/std": 0.15882401168346405, "reward": 0.11840285360813141, "reward_std": 0.15882402658462524, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04550213739275932, "sampling/sampling_logp_difference/max": 1.3136234283447266, "sampling/importance_sampling_ratio/min": 0.2688441574573517, "sampling/importance_sampling_ratio/mean": 1.00211763381958, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.28015392273664474, "clip_ratio/low_mean": 0.01820334349758923, "clip_ratio/low_min": 0.01820334349758923, "clip_ratio/high_mean": 0.011374081019312143, "clip_ratio/high_max": 0.011374081019312143, "clip_ratio/region_mean": 0.029577424516901374, "reward_total_mean": 0.11840285360813141, "reward_meter_mean": 0.11840285360813141, "reward_meter_std": 0.15882401168346405, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.11840285360813141, "reward_total_composite_std": 0.15882401168346405} {"timestamp_utc": "2026-04-12T00:06:37Z", "mode": "train", "global_step": 1391, "epoch": 0.05587018516287103, "loss": -0.0043, "grad_norm": 5.131795406341553, "learning_rate": 5.787878787878788e-06, "num_tokens": 3135632.0, "completions/mean_length": 76.25, "completions/min_length": 71.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.25, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.9707291722297668, "rewards/meter/std": 0.025657914578914642, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9707291722297668, "rewards/total_composite/std": 0.025657914578914642, "reward": 0.9707291722297668, "reward_std": 0.02565791644155979, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0554119348526001, "sampling/sampling_logp_difference/max": 1.9454677104949951, "sampling/importance_sampling_ratio/min": 0.1429203599691391, "sampling/importance_sampling_ratio/mean": 0.9987998008728027, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3864719457924366, "clip_ratio/low_mean": 0.005090707214549184, "clip_ratio/low_min": 0.005090707214549184, "clip_ratio/high_mean": 0.04821625351905823, "clip_ratio/high_max": 0.04821625351905823, "clip_ratio/region_mean": 0.05330696073360741, "reward_total_mean": 0.9707291722297668, "reward_meter_mean": 0.9707291722297668, "reward_meter_std": 0.025657914578914642, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9707291722297668, "reward_total_composite_std": 0.025657914578914642} {"timestamp_utc": "2026-04-12T00:06:42Z", "mode": "train", "global_step": 1392, "epoch": 0.055910350644655986, "loss": 0.0218, "grad_norm": 11.628338813781738, "learning_rate": 5.784848484848486e-06, "num_tokens": 3137481.0, "completions/mean_length": 65.125, "completions/min_length": 63.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.125, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9977819919586182, "rewards/meter/std": 0.0022414319682866335, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977819919586182, "rewards/total_composite/std": 0.0022414319682866335, "reward": 0.9977819919586182, "reward_std": 0.0022414233535528183, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.035027600824832916, "sampling/sampling_logp_difference/max": 0.8262596130371094, "sampling/importance_sampling_ratio/min": 0.4376833438873291, "sampling/importance_sampling_ratio/mean": 1.0126001834869385, "sampling/importance_sampling_ratio/max": 1.7962960004806519, "entropy": 0.2185972724109888, "clip_ratio/low_mean": 0.0036764706019312143, "clip_ratio/low_min": 0.0036764706019312143, "clip_ratio/high_mean": 0.041855036513879895, "clip_ratio/high_max": 0.041855036513879895, "clip_ratio/region_mean": 0.04553150711581111, "reward_total_mean": 0.9977819919586182, "reward_meter_mean": 0.9977819919586182, "reward_meter_std": 0.0022414319682866335, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9977819919586182, "reward_total_composite_std": 0.0022414319682866335} {"timestamp_utc": "2026-04-12T00:06:46Z", "mode": "train", "global_step": 1393, "epoch": 0.05595051612644094, "loss": -0.0412, "grad_norm": 8.172205924987793, "learning_rate": 5.781818181818181e-06, "num_tokens": 3139228.0, "completions/mean_length": 58.375, "completions/min_length": 46.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.375, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.48125261068344116, "rewards/meter/std": 0.2949608266353607, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.45941346883773804, "rewards/total_composite/std": 0.29778623580932617, "reward": 0.45941346883773804, "reward_std": 0.29778623580932617, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07994799315929413, "sampling/sampling_logp_difference/max": 1.2517375946044922, "sampling/importance_sampling_ratio/min": 0.2860074043273926, "sampling/importance_sampling_ratio/mean": 1.0176113843917847, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6722332537174225, "clip_ratio/low_mean": 0.03256637742742896, "clip_ratio/low_min": 0.03256637742742896, "clip_ratio/high_mean": 0.032872630283236504, "clip_ratio/high_max": 0.032872630283236504, "clip_ratio/region_mean": 0.06543900771066546, "reward_total_mean": 0.45941346883773804, "reward_meter_mean": 0.48125261068344116, "reward_meter_std": 0.2949608266353607, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.45941346883773804, "reward_total_composite_std": 0.29778623580932617} {"timestamp_utc": "2026-04-12T00:06:51Z", "mode": "train", "global_step": 1394, "epoch": 0.055990681608225894, "loss": 0.1048, "grad_norm": 5.719612121582031, "learning_rate": 5.7787878787878795e-06, "num_tokens": 3141241.0, "completions/mean_length": 75.625, "completions/min_length": 66.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.625, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.5849511027336121, "rewards/meter/std": 0.4843460023403168, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.824999988079071, "rewards/repeat_penalty/std": 0.19820624589920044, "rewards/total_composite/mean": 0.4432606101036072, "rewards/total_composite/std": 0.39577049016952515, "reward": 0.4432606101036072, "reward_std": 0.39577043056488037, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023130640387535095, "sampling/sampling_logp_difference/max": 0.8899397850036621, "sampling/importance_sampling_ratio/min": 0.41068050265312195, "sampling/importance_sampling_ratio/mean": 1.0027085542678833, "sampling/importance_sampling_ratio/max": 1.5665262937545776, "entropy": 0.136459331959486, "clip_ratio/low_mean": 0.011919185984879732, "clip_ratio/low_min": 0.011919185984879732, "clip_ratio/high_mean": 0.01639902195893228, "clip_ratio/high_max": 0.01639902195893228, "clip_ratio/region_mean": 0.028318207943812013, "reward_total_mean": 0.4432606101036072, "reward_meter_mean": 0.5849511027336121, "reward_meter_std": 0.4843460023403168, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.824999988079071, "reward_repeat_penalty_std": 0.19820624589920044, "reward_total_composite_mean": 0.4432606101036072, "reward_total_composite_std": 0.39577049016952515} {"timestamp_utc": "2026-04-12T00:06:56Z", "mode": "train", "global_step": 1395, "epoch": 0.05603084709001085, "loss": 0.0056, "grad_norm": 4.375842094421387, "learning_rate": 5.775757575757577e-06, "num_tokens": 3143152.0, "completions/mean_length": 69.875, "completions/min_length": 64.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.875, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9944968223571777, "rewards/meter/std": 0.0024136791471391916, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9529762864112854, "rewards/total_composite/std": 0.11665359884500504, "reward": 0.9529762864112854, "reward_std": 0.11665360629558563, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04802675172686577, "sampling/sampling_logp_difference/max": 1.7336769104003906, "sampling/importance_sampling_ratio/min": 0.17663374543190002, "sampling/importance_sampling_ratio/mean": 1.009644627571106, "sampling/importance_sampling_ratio/max": 1.78848135471344, "entropy": 0.37176400050520897, "clip_ratio/low_mean": 0.0017857142956927419, "clip_ratio/low_min": 0.0017857142956927419, "clip_ratio/high_mean": 0.03542683261912316, "clip_ratio/high_max": 0.03542683261912316, "clip_ratio/region_mean": 0.0372125469148159, "reward_total_mean": 0.9529762864112854, "reward_meter_mean": 0.9944968223571777, "reward_meter_std": 0.0024136791471391916, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.9529762864112854, "reward_total_composite_std": 0.11665359884500504} {"timestamp_utc": "2026-04-12T00:07:04Z", "mode": "train", "global_step": 1396, "epoch": 0.0560710125717958, "loss": -0.0037, "grad_norm": 2.749262809753418, "learning_rate": 5.772727272727273e-06, "num_tokens": 3147511.0, "completions/mean_length": 316.875, "completions/min_length": 296.0, "completions/max_length": 346.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 316.875, "completions/min_terminated_length": 296.0, "completions/max_terminated_length": 346.0, "rewards/meter/mean": 0.323222815990448, "rewards/meter/std": 0.21111111342906952, "rewards/count_adherence/mean": 0.6111111044883728, "rewards/count_adherence/std": 0.029695691540837288, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7082297801971436, "rewards/repeat_penalty/std": 0.0814073234796524, "rewards/total_composite/mean": 0.13397403061389923, "rewards/total_composite/std": 0.07767514884471893, "reward": 0.13397403061389923, "reward_std": 0.07767514139413834, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027990715578198433, "sampling/sampling_logp_difference/max": 4.105655193328857, "sampling/importance_sampling_ratio/min": 0.016479218378663063, "sampling/importance_sampling_ratio/mean": 0.9997315406799316, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0825919434428215, "clip_ratio/low_mean": 0.010312073398381472, "clip_ratio/low_min": 0.010312073398381472, "clip_ratio/high_mean": 0.012935725739225745, "clip_ratio/high_max": 0.012935725739225745, "clip_ratio/region_mean": 0.023247799137607217, "reward_total_mean": 0.13397403061389923, "reward_meter_mean": 0.323222815990448, "reward_meter_std": 0.21111111342906952, "reward_count_adherence_mean": 0.6111111044883728, "reward_count_adherence_std": 0.029695691540837288, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7082297801971436, "reward_repeat_penalty_std": 0.0814073234796524, "reward_total_composite_mean": 0.13397403061389923, "reward_total_composite_std": 0.07767514884471893} {"timestamp_utc": "2026-04-12T00:07:08Z", "mode": "train", "global_step": 1397, "epoch": 0.056111178053580756, "loss": 0.0166, "grad_norm": 4.050926208496094, "learning_rate": 5.76969696969697e-06, "num_tokens": 3149369.0, "completions/mean_length": 70.25, "completions/min_length": 66.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.25, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9413079023361206, "rewards/meter/std": 0.009388444945216179, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.753129243850708, "rewards/total_composite/std": 0.10083267837762833, "reward": 0.753129243850708, "reward_std": 0.10083267837762833, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018803365528583527, "sampling/sampling_logp_difference/max": 0.8203344345092773, "sampling/importance_sampling_ratio/min": 0.44028440117836, "sampling/importance_sampling_ratio/mean": 0.9976539611816406, "sampling/importance_sampling_ratio/max": 1.494532585144043, "entropy": 0.08694514073431492, "clip_ratio/low_mean": 0.0034246575087308884, "clip_ratio/low_min": 0.0034246575087308884, "clip_ratio/high_mean": 0.00919352809432894, "clip_ratio/high_max": 0.00919352809432894, "clip_ratio/region_mean": 0.012618185603059828, "reward_total_mean": 0.753129243850708, "reward_meter_mean": 0.9413079023361206, "reward_meter_std": 0.009388444945216179, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.10690449178218842, "reward_total_composite_mean": 0.753129243850708, "reward_total_composite_std": 0.10083267837762833} {"timestamp_utc": "2026-04-12T00:07:13Z", "mode": "train", "global_step": 1398, "epoch": 0.05615134353536571, "loss": -0.0167, "grad_norm": 4.091942310333252, "learning_rate": 5.766666666666667e-06, "num_tokens": 3151224.0, "completions/mean_length": 58.875, "completions/min_length": 54.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.875, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.6508887410163879, "rewards/meter/std": 0.1606731116771698, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6508887410163879, "rewards/total_composite/std": 0.1606731116771698, "reward": 0.6508887410163879, "reward_std": 0.1606731414794922, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02822190895676613, "sampling/sampling_logp_difference/max": 0.8628159761428833, "sampling/importance_sampling_ratio/min": 0.4910596013069153, "sampling/importance_sampling_ratio/mean": 1.0124748945236206, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17077440861612558, "clip_ratio/low_mean": 0.012992122676223516, "clip_ratio/low_min": 0.012992122676223516, "clip_ratio/high_mean": 0.0061509020160883665, "clip_ratio/high_max": 0.0061509020160883665, "clip_ratio/region_mean": 0.019143024692311883, "reward_total_mean": 0.6508887410163879, "reward_meter_mean": 0.6508887410163879, "reward_meter_std": 0.1606731116771698, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6508887410163879, "reward_total_composite_std": 0.1606731116771698} {"timestamp_utc": "2026-04-12T00:07:17Z", "mode": "train", "global_step": 1399, "epoch": 0.05619150901715066, "loss": 0.0332, "grad_norm": 7.159199237823486, "learning_rate": 5.763636363636365e-06, "num_tokens": 3152918.0, "completions/mean_length": 63.75, "completions/min_length": 61.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.75, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9983058571815491, "rewards/meter/std": 0.00029361830092966557, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9983058571815491, "rewards/total_composite/std": 0.00029361830092966557, "reward": 0.9983058571815491, "reward_std": 0.000293627759674564, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03937718644738197, "sampling/sampling_logp_difference/max": 2.0187900066375732, "sampling/importance_sampling_ratio/min": 0.13281607627868652, "sampling/importance_sampling_ratio/mean": 1.0061930418014526, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1746031977236271, "clip_ratio/low_mean": 0.00562528264708817, "clip_ratio/low_min": 0.00562528264708817, "clip_ratio/high_mean": 0.011937764240428805, "clip_ratio/high_max": 0.011937764240428805, "clip_ratio/region_mean": 0.017563046887516975, "reward_total_mean": 0.9983058571815491, "reward_meter_mean": 0.9983058571815491, "reward_meter_std": 0.00029361830092966557, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9983058571815491, "reward_total_composite_std": 0.00029361830092966557} {"timestamp_utc": "2026-04-12T00:07:23Z", "mode": "train", "global_step": 1400, "epoch": 0.05623167449893562, "loss": 0.0063, "grad_norm": 3.5979604721069336, "learning_rate": 5.760606060606061e-06, "num_tokens": 3155594.0, "completions/mean_length": 165.5, "completions/min_length": 159.0, "completions/max_length": 171.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 165.5, "completions/min_terminated_length": 159.0, "completions/max_terminated_length": 171.0, "rewards/meter/mean": 0.8759403228759766, "rewards/meter/std": 0.19313931465148926, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9027777910232544, "rewards/repeat_penalty/std": 0.0927247703075409, "rewards/total_composite/mean": 0.7801923751831055, "rewards/total_composite/std": 0.1476370245218277, "reward": 0.7801923751831055, "reward_std": 0.1476370245218277, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05515680089592934, "sampling/sampling_logp_difference/max": 1.6768007278442383, "sampling/importance_sampling_ratio/min": 0.18697118759155273, "sampling/importance_sampling_ratio/mean": 1.0012664794921875, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4123094528913498, "clip_ratio/low_mean": 0.022338663460686803, "clip_ratio/low_min": 0.022338663460686803, "clip_ratio/high_mean": 0.027914301492273808, "clip_ratio/high_max": 0.027914301492273808, "clip_ratio/region_mean": 0.05025296495296061, "reward_total_mean": 0.7801923751831055, "reward_meter_mean": 0.8759403228759766, "reward_meter_std": 0.19313931465148926, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9027777910232544, "reward_repeat_penalty_std": 0.0927247703075409, "reward_total_composite_mean": 0.7801923751831055, "reward_total_composite_std": 0.1476370245218277} {"timestamp_utc": "2026-04-12T00:08:18Z", "mode": "eval", "global_step": 1400, "epoch": 0.05623167449893562, "eval_loss": NaN, "eval_runtime": 55.0853, "eval_samples_per_second": 1.888, "eval_steps_per_second": 0.236, "eval_num_tokens": 3155594.0, "eval_completions/mean_length": 158.23076923076923, "eval_completions/min_length": 57.92307692307692, "eval_completions/max_length": 281.6923076923077, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 158.23076923076923, "eval_completions/min_terminated_length": 57.92307692307692, "eval_completions/max_terminated_length": 281.6923076923077, "eval_rewards/meter/mean": 0.6238016211069547, "eval_rewards/meter/std": 0.41127324333557713, "eval_rewards/count_adherence/mean": 0.8451297649970422, "eval_rewards/count_adherence/std": 0.1567458017514302, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/repeat_penalty/mean": 0.7824494563616239, "eval_rewards/repeat_penalty/std": 0.1815235666357554, "eval_rewards/total_composite/mean": 0.42582815427046555, "eval_rewards/total_composite/std": 0.3518183070879716, "eval_reward": 0.42582815427046555, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.02079938738965071, "eval_sampling/sampling_logp_difference/max": 0.9362153823559101, "eval_sampling/importance_sampling_ratio/min": 0.40005409259062547, "eval_sampling/importance_sampling_ratio/mean": 1.0046258247815645, "eval_sampling/importance_sampling_ratio/max": 1.4843606765453632, "eval_entropy": 0.18456935252134615, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.42582815427046555, "eval_reward_meter_mean": 0.6238016211069547, "eval_reward_meter_std": 0.41127324333557713, "eval_reward_count_adherence_mean": 0.8451297649970422, "eval_reward_count_adherence_std": 0.1567458017514302, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_repeat_penalty_mean": 0.7824494563616239, "eval_reward_repeat_penalty_std": 0.1815235666357554, "eval_reward_total_composite_mean": 0.42582815427046555, "eval_reward_total_composite_std": 0.3518183070879716} {"timestamp_utc": "2026-04-12T00:08:25Z", "mode": "train", "global_step": 1401, "epoch": 0.05627183998072057, "loss": 0.0053, "grad_norm": 2.925625801086426, "learning_rate": 5.7575757575757586e-06, "num_tokens": 3157247.0, "completions/mean_length": 61.625, "completions/min_length": 61.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.625, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9903584122657776, "rewards/meter/std": 0.011271141469478607, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9903584122657776, "rewards/total_composite/std": 0.011271141469478607, "reward": 0.9903584122657776, "reward_std": 0.011271136812865734, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015541310422122478, "sampling/sampling_logp_difference/max": 1.4758706092834473, "sampling/importance_sampling_ratio/min": 0.22857964038848877, "sampling/importance_sampling_ratio/mean": 1.0016825199127197, "sampling/importance_sampling_ratio/max": 1.5843480825424194, "entropy": 0.07412398885935545, "clip_ratio/low_mean": 0.004032257944345474, "clip_ratio/low_min": 0.004032257944345474, "clip_ratio/high_mean": 0.014245107769966125, "clip_ratio/high_max": 0.014245107769966125, "clip_ratio/region_mean": 0.0182773657143116, "reward_total_mean": 0.9903584122657776, "reward_meter_mean": 0.9903584122657776, "reward_meter_std": 0.011271141469478607, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9903584122657776, "reward_total_composite_std": 0.011271141469478607} {"timestamp_utc": "2026-04-12T00:08:31Z", "mode": "train", "global_step": 1402, "epoch": 0.056312005462505525, "loss": -0.0084, "grad_norm": 2.7992911338806152, "learning_rate": 5.754545454545455e-06, "num_tokens": 3160271.0, "completions/mean_length": 139.0, "completions/min_length": 133.0, "completions/max_length": 147.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 139.0, "completions/min_terminated_length": 133.0, "completions/max_terminated_length": 147.0, "rewards/meter/mean": 0.4930988550186157, "rewards/meter/std": 0.10791701823472977, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4305555522441864, "rewards/repeat_penalty/std": 0.15068919956684113, "rewards/total_composite/mean": 0.1747434139251709, "rewards/total_composite/std": 0.07833553850650787, "reward": 0.1747434139251709, "reward_std": 0.07833553850650787, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02746950089931488, "sampling/sampling_logp_difference/max": 1.4570236206054688, "sampling/importance_sampling_ratio/min": 0.23292852938175201, "sampling/importance_sampling_ratio/mean": 1.0046955347061157, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15692214109003544, "clip_ratio/low_mean": 0.013523990812245756, "clip_ratio/low_min": 0.013523990812245756, "clip_ratio/high_mean": 0.0025862068869173527, "clip_ratio/high_max": 0.0025862068869173527, "clip_ratio/region_mean": 0.01611019769916311, "reward_total_mean": 0.1747434139251709, "reward_meter_mean": 0.4930988550186157, "reward_meter_std": 0.10791701823472977, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4305555522441864, "reward_repeat_penalty_std": 0.15068919956684113, "reward_total_composite_mean": 0.1747434139251709, "reward_total_composite_std": 0.07833553850650787} {"timestamp_utc": "2026-04-12T00:08:37Z", "mode": "train", "global_step": 1403, "epoch": 0.05635217094429048, "loss": -0.0093, "grad_norm": 3.119965076446533, "learning_rate": 5.751515151515152e-06, "num_tokens": 3163193.0, "completions/mean_length": 187.25, "completions/min_length": 157.0, "completions/max_length": 202.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 187.25, "completions/min_terminated_length": 157.0, "completions/max_terminated_length": 202.0, "rewards/meter/mean": 0.958865761756897, "rewards/meter/std": 0.024461213499307632, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.829365074634552, "rewards/repeat_penalty/std": 0.10101525485515594, "rewards/total_composite/mean": 0.7755063772201538, "rewards/total_composite/std": 0.11600048094987869, "reward": 0.7755063772201538, "reward_std": 0.1160004660487175, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030590685084462166, "sampling/sampling_logp_difference/max": 1.2629318237304688, "sampling/importance_sampling_ratio/min": 0.2828236222267151, "sampling/importance_sampling_ratio/mean": 1.0046159029006958, "sampling/importance_sampling_ratio/max": 1.7141207456588745, "entropy": 0.19072475843131542, "clip_ratio/low_mean": 0.0072619569837115705, "clip_ratio/low_min": 0.0072619569837115705, "clip_ratio/high_mean": 0.007620060350745916, "clip_ratio/high_max": 0.007620060350745916, "clip_ratio/region_mean": 0.014882017334457487, "reward_total_mean": 0.7755063772201538, "reward_meter_mean": 0.958865761756897, "reward_meter_std": 0.024461213499307632, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.829365074634552, "reward_repeat_penalty_std": 0.10101525485515594, "reward_total_composite_mean": 0.7755063772201538, "reward_total_composite_std": 0.11600048094987869} {"timestamp_utc": "2026-04-12T00:08:42Z", "mode": "train", "global_step": 1404, "epoch": 0.05639233642607543, "loss": 0.0225, "grad_norm": 4.241501808166504, "learning_rate": 5.748484848484849e-06, "num_tokens": 3165135.0, "completions/mean_length": 77.75, "completions/min_length": 74.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.75, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.9753485321998596, "rewards/meter/std": 0.02492102049291134, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9753485321998596, "rewards/total_composite/std": 0.02492102049291134, "reward": 0.9753485321998596, "reward_std": 0.024921026080846786, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04453163966536522, "sampling/sampling_logp_difference/max": 1.5494651794433594, "sampling/importance_sampling_ratio/min": 0.21236151456832886, "sampling/importance_sampling_ratio/mean": 1.0083844661712646, "sampling/importance_sampling_ratio/max": 1.8341128826141357, "entropy": 0.3366150539368391, "clip_ratio/low_mean": 0.0077174786711111665, "clip_ratio/low_min": 0.0077174786711111665, "clip_ratio/high_mean": 0.019914651405997574, "clip_ratio/high_max": 0.019914651405997574, "clip_ratio/region_mean": 0.02763213007710874, "reward_total_mean": 0.9753485321998596, "reward_meter_mean": 0.9753485321998596, "reward_meter_std": 0.02492102049291134, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9753485321998596, "reward_total_composite_std": 0.02492102049291134} {"timestamp_utc": "2026-04-12T00:08:47Z", "mode": "train", "global_step": 1405, "epoch": 0.05643250190786039, "loss": 0.0037, "grad_norm": 3.760753631591797, "learning_rate": 5.745454545454546e-06, "num_tokens": 3166898.0, "completions/mean_length": 67.375, "completions/min_length": 65.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.375, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9491584897041321, "rewards/meter/std": 0.09555412083864212, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9491584897041321, "rewards/total_composite/std": 0.09555412083864212, "reward": 0.9491584897041321, "reward_std": 0.09555413573980331, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026108134537935257, "sampling/sampling_logp_difference/max": 1.3400421142578125, "sampling/importance_sampling_ratio/min": 0.26183465123176575, "sampling/importance_sampling_ratio/mean": 1.005703091621399, "sampling/importance_sampling_ratio/max": 1.8443472385406494, "entropy": 0.20304011879488826, "clip_ratio/low_mean": 0.0018382353009656072, "clip_ratio/low_min": 0.0018382353009656072, "clip_ratio/high_mean": 0.0202326342696324, "clip_ratio/high_max": 0.0202326342696324, "clip_ratio/region_mean": 0.022070869570598006, "reward_total_mean": 0.9491584897041321, "reward_meter_mean": 0.9491584897041321, "reward_meter_std": 0.09555412083864212, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9491584897041321, "reward_total_composite_std": 0.09555412083864212} {"timestamp_utc": "2026-04-12T00:08:52Z", "mode": "train", "global_step": 1406, "epoch": 0.05647266738964534, "loss": 0.0102, "grad_norm": 5.033778190612793, "learning_rate": 5.742424242424242e-06, "num_tokens": 3168628.0, "completions/mean_length": 59.25, "completions/min_length": 55.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.25, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.37988948822021484, "rewards/meter/std": 0.33830785751342773, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.37988948822021484, "rewards/total_composite/std": 0.33830785751342773, "reward": 0.37988948822021484, "reward_std": 0.33830785751342773, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05973369628190994, "sampling/sampling_logp_difference/max": 1.5036615133285522, "sampling/importance_sampling_ratio/min": 0.22231467068195343, "sampling/importance_sampling_ratio/mean": 1.01566481590271, "sampling/importance_sampling_ratio/max": 1.7001794576644897, "entropy": 0.4192809797823429, "clip_ratio/low_mean": 0.025304300244897604, "clip_ratio/low_min": 0.025304300244897604, "clip_ratio/high_mean": 0.03157911077141762, "clip_ratio/high_max": 0.03157911077141762, "clip_ratio/region_mean": 0.05688341101631522, "reward_total_mean": 0.37988948822021484, "reward_meter_mean": 0.37988948822021484, "reward_meter_std": 0.33830785751342773, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.37988948822021484, "reward_total_composite_std": 0.33830785751342773} {"timestamp_utc": "2026-04-12T00:08:57Z", "mode": "train", "global_step": 1407, "epoch": 0.056512832871430295, "loss": 0.0163, "grad_norm": 6.2097625732421875, "learning_rate": 5.73939393939394e-06, "num_tokens": 3170743.0, "completions/mean_length": 99.375, "completions/min_length": 94.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.375, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.8928050398826599, "rewards/meter/std": 0.15822991728782654, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9249999523162842, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.829431414604187, "rewards/total_composite/std": 0.1831192672252655, "reward": 0.829431414604187, "reward_std": 0.1831192672252655, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04797036573290825, "sampling/sampling_logp_difference/max": 1.2440190315246582, "sampling/importance_sampling_ratio/min": 0.2882235050201416, "sampling/importance_sampling_ratio/mean": 1.0000447034835815, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.32133867032825947, "clip_ratio/low_mean": 0.01607399620115757, "clip_ratio/low_min": 0.01607399620115757, "clip_ratio/high_mean": 0.0139403420034796, "clip_ratio/high_max": 0.0139403420034796, "clip_ratio/region_mean": 0.03001433820463717, "reward_total_mean": 0.829431414604187, "reward_meter_mean": 0.8928050398826599, "reward_meter_std": 0.15822991728782654, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9249999523162842, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.829431414604187, "reward_total_composite_std": 0.1831192672252655} {"timestamp_utc": "2026-04-12T00:09:01Z", "mode": "train", "global_step": 1408, "epoch": 0.05655299835321525, "loss": -0.0044, "grad_norm": 7.772467136383057, "learning_rate": 5.736363636363637e-06, "num_tokens": 3172326.0, "completions/mean_length": 30.875, "completions/min_length": 30.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.875, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9951827526092529, "rewards/meter/std": 0.007008199580013752, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9951827526092529, "rewards/total_composite/std": 0.007008199580013752, "reward": 0.9951827526092529, "reward_std": 0.007008187007158995, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.033370744436979294, "sampling/sampling_logp_difference/max": 0.8426570892333984, "sampling/importance_sampling_ratio/min": 0.4305649697780609, "sampling/importance_sampling_ratio/mean": 0.9993618726730347, "sampling/importance_sampling_ratio/max": 1.774673581123352, "entropy": 0.1420626100152731, "clip_ratio/low_mean": 0.008198924828320742, "clip_ratio/low_min": 0.008198924828320742, "clip_ratio/high_mean": 0.01601142482832074, "clip_ratio/high_max": 0.01601142482832074, "clip_ratio/region_mean": 0.024210349656641483, "reward_total_mean": 0.9951827526092529, "reward_meter_mean": 0.9951827526092529, "reward_meter_std": 0.007008199580013752, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9951827526092529, "reward_total_composite_std": 0.007008199580013752} {"timestamp_utc": "2026-04-12T00:09:05Z", "mode": "train", "global_step": 1409, "epoch": 0.0565931638350002, "loss": 0.2629, "grad_norm": 11.023374557495117, "learning_rate": 5.733333333333334e-06, "num_tokens": 3173907.0, "completions/mean_length": 51.625, "completions/min_length": 36.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.625, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.992419958114624, "rewards/meter/std": 0.00853477232158184, "rewards/count_adherence/mean": 0.5, "rewards/count_adherence/std": 0.5345224738121033, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4936572313308716, "rewards/total_composite/std": 0.5277820229530334, "reward": 0.4936572313308716, "reward_std": 0.5277819633483887, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06906002014875412, "sampling/sampling_logp_difference/max": 3.973472833633423, "sampling/importance_sampling_ratio/min": 0.018808001652359962, "sampling/importance_sampling_ratio/mean": 0.9999293088912964, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.28332205303013325, "clip_ratio/low_mean": 0.02108370151836425, "clip_ratio/low_min": 0.02108370151836425, "clip_ratio/high_mean": 0.016995614394545555, "clip_ratio/high_max": 0.016995614394545555, "clip_ratio/region_mean": 0.038079315912909806, "reward_total_mean": 0.4936572313308716, "reward_meter_mean": 0.992419958114624, "reward_meter_std": 0.00853477232158184, "reward_count_adherence_mean": 0.5, "reward_count_adherence_std": 0.5345224738121033, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.4936572313308716, "reward_total_composite_std": 0.5277820229530334} {"timestamp_utc": "2026-04-12T00:09:10Z", "mode": "train", "global_step": 1410, "epoch": 0.05663332931678516, "loss": 0.0359, "grad_norm": 5.875049591064453, "learning_rate": 5.7303030303030305e-06, "num_tokens": 3175555.0, "completions/mean_length": 58.0, "completions/min_length": 54.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.7323773503303528, "rewards/meter/std": 0.34669190645217896, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7323773503303528, "rewards/total_composite/std": 0.34669190645217896, "reward": 0.7323773503303528, "reward_std": 0.34669187664985657, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04662606492638588, "sampling/sampling_logp_difference/max": 2.092195987701416, "sampling/importance_sampling_ratio/min": 0.12341582030057907, "sampling/importance_sampling_ratio/mean": 1.0006393194198608, "sampling/importance_sampling_ratio/max": 1.7754662036895752, "entropy": 0.20567422546446323, "clip_ratio/low_mean": 0.01024590153247118, "clip_ratio/low_min": 0.01024590153247118, "clip_ratio/high_mean": 0.04356799623928964, "clip_ratio/high_max": 0.04356799623928964, "clip_ratio/region_mean": 0.05381389777176082, "reward_total_mean": 0.7323773503303528, "reward_meter_mean": 0.7323773503303528, "reward_meter_std": 0.34669190645217896, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7323773503303528, "reward_total_composite_std": 0.34669190645217896} {"timestamp_utc": "2026-04-12T00:09:15Z", "mode": "train", "global_step": 1411, "epoch": 0.05667349479857011, "loss": 0.0404, "grad_norm": 7.019533157348633, "learning_rate": 5.727272727272728e-06, "num_tokens": 3177472.0, "completions/mean_length": 58.625, "completions/min_length": 54.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.625, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.930023729801178, "rewards/meter/std": 0.11754162609577179, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.930023729801178, "rewards/total_composite/std": 0.11754162609577179, "reward": 0.930023729801178, "reward_std": 0.1175416111946106, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03719957172870636, "sampling/sampling_logp_difference/max": 0.9408185482025146, "sampling/importance_sampling_ratio/min": 0.3903082311153412, "sampling/importance_sampling_ratio/mean": 1.0033373832702637, "sampling/importance_sampling_ratio/max": 1.9308364391326904, "entropy": 0.22591773979365826, "clip_ratio/low_mean": 0.007942708441987634, "clip_ratio/low_min": 0.007942708441987634, "clip_ratio/high_mean": 0.02417309512384236, "clip_ratio/high_max": 0.02417309512384236, "clip_ratio/region_mean": 0.03211580356582999, "reward_total_mean": 0.930023729801178, "reward_meter_mean": 0.930023729801178, "reward_meter_std": 0.11754162609577179, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.930023729801178, "reward_total_composite_std": 0.11754162609577179} {"timestamp_utc": "2026-04-12T00:09:20Z", "mode": "train", "global_step": 1412, "epoch": 0.056713660280355065, "loss": 0.0095, "grad_norm": 3.1584293842315674, "learning_rate": 5.724242424242424e-06, "num_tokens": 3179678.0, "completions/mean_length": 121.75, "completions/min_length": 119.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 121.75, "completions/min_terminated_length": 119.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.980501115322113, "rewards/meter/std": 0.04536837711930275, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.8404378890991211, "rewards/total_composite/std": 0.0856878012418747, "reward": 0.8404378890991211, "reward_std": 0.0856877937912941, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04456208273768425, "sampling/sampling_logp_difference/max": 1.8470945358276367, "sampling/importance_sampling_ratio/min": 0.19929902255535126, "sampling/importance_sampling_ratio/mean": 1.0000073909759521, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.25815106742084026, "clip_ratio/low_mean": 0.007936508394777775, "clip_ratio/low_min": 0.007936508394777775, "clip_ratio/high_mean": 0.03802949539385736, "clip_ratio/high_max": 0.03802949539385736, "clip_ratio/region_mean": 0.045966003788635135, "reward_total_mean": 0.8404378890991211, "reward_meter_mean": 0.980501115322113, "reward_meter_std": 0.04536837711930275, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.07636035233736038, "reward_total_composite_mean": 0.8404378890991211, "reward_total_composite_std": 0.0856878012418747} {"timestamp_utc": "2026-04-12T00:09:26Z", "mode": "train", "global_step": 1413, "epoch": 0.05675382576214002, "loss": -0.0061, "grad_norm": 2.6527252197265625, "learning_rate": 5.721212121212122e-06, "num_tokens": 3182168.0, "completions/mean_length": 148.25, "completions/min_length": 143.0, "completions/max_length": 155.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 148.25, "completions/min_terminated_length": 143.0, "completions/max_terminated_length": 155.0, "rewards/meter/mean": 0.9954190850257874, "rewards/meter/std": 0.002940161619335413, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.8531917333602905, "rewards/total_composite/std": 0.07582316547632217, "reward": 0.8531917333602905, "reward_std": 0.07582316547632217, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04073692858219147, "sampling/sampling_logp_difference/max": 1.4822022914886475, "sampling/importance_sampling_ratio/min": 0.22713692486286163, "sampling/importance_sampling_ratio/mean": 1.0049986839294434, "sampling/importance_sampling_ratio/max": 1.6937066316604614, "entropy": 0.24800713174045086, "clip_ratio/low_mean": 0.02131642634049058, "clip_ratio/low_min": 0.02131642634049058, "clip_ratio/high_mean": 0.019891314674168825, "clip_ratio/high_max": 0.019891314674168825, "clip_ratio/region_mean": 0.041207741014659405, "reward_total_mean": 0.8531917333602905, "reward_meter_mean": 0.9954190850257874, "reward_meter_std": 0.002940161619335413, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.07636035233736038, "reward_total_composite_mean": 0.8531917333602905, "reward_total_composite_std": 0.07582316547632217} {"timestamp_utc": "2026-04-12T00:09:30Z", "mode": "train", "global_step": 1414, "epoch": 0.05679399124392497, "loss": -0.0031, "grad_norm": 8.496710777282715, "learning_rate": 5.718181818181819e-06, "num_tokens": 3183867.0, "completions/mean_length": 66.375, "completions/min_length": 64.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.375, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.7868467569351196, "rewards/meter/std": 0.3379902243614197, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7868467569351196, "rewards/total_composite/std": 0.3379902243614197, "reward": 0.7868467569351196, "reward_std": 0.3379901945590973, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03707931190729141, "sampling/sampling_logp_difference/max": 1.9779717922210693, "sampling/importance_sampling_ratio/min": 0.13834954798221588, "sampling/importance_sampling_ratio/mean": 0.9952957034111023, "sampling/importance_sampling_ratio/max": 1.4725927114486694, "entropy": 0.16014890000224113, "clip_ratio/low_mean": 0.007663170341402292, "clip_ratio/low_min": 0.007663170341402292, "clip_ratio/high_mean": 0.0170001951046288, "clip_ratio/high_max": 0.0170001951046288, "clip_ratio/region_mean": 0.024663365446031094, "reward_total_mean": 0.7868467569351196, "reward_meter_mean": 0.7868467569351196, "reward_meter_std": 0.3379902243614197, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7868467569351196, "reward_total_composite_std": 0.3379902243614197} {"timestamp_utc": "2026-04-12T00:09:35Z", "mode": "train", "global_step": 1415, "epoch": 0.056834156725709926, "loss": 0.0077, "grad_norm": 12.028426170349121, "learning_rate": 5.715151515151516e-06, "num_tokens": 3185507.0, "completions/mean_length": 41.0, "completions/min_length": 40.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.0, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.9893122911453247, "rewards/meter/std": 0.020975705236196518, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9893122911453247, "rewards/total_composite/std": 0.020975705236196518, "reward": 0.9893122911453247, "reward_std": 0.020975695922970772, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07620865851640701, "sampling/sampling_logp_difference/max": 2.0691659450531006, "sampling/importance_sampling_ratio/min": 0.12629108130931854, "sampling/importance_sampling_ratio/mean": 1.0033856630325317, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4230155944824219, "clip_ratio/low_mean": 0.00937500037252903, "clip_ratio/low_min": 0.00937500037252903, "clip_ratio/high_mean": 0.04510642075911164, "clip_ratio/high_max": 0.04510642075911164, "clip_ratio/region_mean": 0.05448142113164067, "reward_total_mean": 0.9893122911453247, "reward_meter_mean": 0.9893122911453247, "reward_meter_std": 0.020975705236196518, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9893122911453247, "reward_total_composite_std": 0.020975705236196518} {"timestamp_utc": "2026-04-12T00:09:41Z", "mode": "train", "global_step": 1416, "epoch": 0.05687432220749488, "loss": 0.0021, "grad_norm": 1.9633679389953613, "learning_rate": 5.712121212121212e-06, "num_tokens": 3188624.0, "completions/mean_length": 179.625, "completions/min_length": 177.0, "completions/max_length": 183.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 179.625, "completions/min_terminated_length": 177.0, "completions/max_terminated_length": 183.0, "rewards/meter/mean": 0.9946449995040894, "rewards/meter/std": 0.006311771925538778, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8055555820465088, "rewards/repeat_penalty/std": 0.07856741547584534, "rewards/total_composite/mean": 0.8011291027069092, "rewards/total_composite/std": 0.07706641405820847, "reward": 0.8011291027069092, "reward_std": 0.07706642150878906, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025626802816987038, "sampling/sampling_logp_difference/max": 1.5970072746276855, "sampling/importance_sampling_ratio/min": 0.20250163972377777, "sampling/importance_sampling_ratio/mean": 1.0057470798492432, "sampling/importance_sampling_ratio/max": 1.6300792694091797, "entropy": 0.15431117359548807, "clip_ratio/low_mean": 0.009044285747222602, "clip_ratio/low_min": 0.009044285747222602, "clip_ratio/high_mean": 0.006235292763449252, "clip_ratio/high_max": 0.006235292763449252, "clip_ratio/region_mean": 0.015279578510671854, "reward_total_mean": 0.8011291027069092, "reward_meter_mean": 0.9946449995040894, "reward_meter_std": 0.006311771925538778, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8055555820465088, "reward_repeat_penalty_std": 0.07856741547584534, "reward_total_composite_mean": 0.8011291027069092, "reward_total_composite_std": 0.07706641405820847} {"timestamp_utc": "2026-04-12T00:09:45Z", "mode": "train", "global_step": 1417, "epoch": 0.056914487689279834, "loss": -0.0002, "grad_norm": 3.094541549682617, "learning_rate": 5.7090909090909096e-06, "num_tokens": 3190361.0, "completions/mean_length": 61.125, "completions/min_length": 60.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.125, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.998288631439209, "rewards/meter/std": 0.00019117545161861926, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998288631439209, "rewards/total_composite/std": 0.00019117545161861926, "reward": 0.998288631439209, "reward_std": 0.00019117545161861926, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01802898570895195, "sampling/sampling_logp_difference/max": 1.3229293823242188, "sampling/importance_sampling_ratio/min": 0.26635390520095825, "sampling/importance_sampling_ratio/mean": 1.0027981996536255, "sampling/importance_sampling_ratio/max": 1.6339690685272217, "entropy": 0.06280876183882356, "clip_ratio/low_mean": 0.0020833334419876337, "clip_ratio/low_min": 0.0020833334419876337, "clip_ratio/high_mean": 0.004098360426723957, "clip_ratio/high_max": 0.004098360426723957, "clip_ratio/region_mean": 0.006181693868711591, "reward_total_mean": 0.998288631439209, "reward_meter_mean": 0.998288631439209, "reward_meter_std": 0.00019117545161861926, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.998288631439209, "reward_total_composite_std": 0.00019117545161861926} {"timestamp_utc": "2026-04-12T00:09:50Z", "mode": "train", "global_step": 1418, "epoch": 0.05695465317106479, "loss": 0.0502, "grad_norm": 10.003430366516113, "learning_rate": 5.706060606060606e-06, "num_tokens": 3191990.0, "completions/mean_length": 45.625, "completions/min_length": 42.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.625, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.8391948938369751, "rewards/meter/std": 0.1604534387588501, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8391948938369751, "rewards/total_composite/std": 0.1604534387588501, "reward": 0.8391948938369751, "reward_std": 0.1604534238576889, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05166694149374962, "sampling/sampling_logp_difference/max": 1.3240852355957031, "sampling/importance_sampling_ratio/min": 0.2660462260246277, "sampling/importance_sampling_ratio/mean": 1.0050475597381592, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.32396047934889793, "clip_ratio/low_mean": 0.013166894670575857, "clip_ratio/low_min": 0.013166894670575857, "clip_ratio/high_mean": 0.013707729522138834, "clip_ratio/high_max": 0.013707729522138834, "clip_ratio/region_mean": 0.02687462419271469, "reward_total_mean": 0.8391948938369751, "reward_meter_mean": 0.8391948938369751, "reward_meter_std": 0.1604534387588501, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8391948938369751, "reward_total_composite_std": 0.1604534387588501} {"timestamp_utc": "2026-04-12T00:09:54Z", "mode": "train", "global_step": 1419, "epoch": 0.05699481865284974, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.703030303030303e-06, "num_tokens": 3193518.0, "completions/mean_length": 31.0, "completions/min_length": 31.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.9990130662918091, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990130662918091, "rewards/total_composite/std": 0.0, "reward": 0.9990130662918091, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.006462951190769672, "sampling/sampling_logp_difference/max": 0.3100270926952362, "sampling/importance_sampling_ratio/min": 0.733427107334137, "sampling/importance_sampling_ratio/mean": 1.0024038553237915, "sampling/importance_sampling_ratio/max": 1.2404334545135498, "entropy": 0.038966658525168896, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9990130662918091, "reward_meter_mean": 0.9990130662918091, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990130662918091, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T00:10:00Z", "mode": "train", "global_step": 1420, "epoch": 0.057034984134634696, "loss": 0.0034, "grad_norm": 3.3092191219329834, "learning_rate": 5.7e-06, "num_tokens": 3195977.0, "completions/mean_length": 118.375, "completions/min_length": 115.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.375, "completions/min_terminated_length": 115.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.9984818696975708, "rewards/meter/std": 8.360291394637898e-05, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8035714030265808, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.6418830156326294, "rewards/total_composite/std": 0.05907980352640152, "reward": 0.6418830156326294, "reward_std": 0.05907979980111122, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015416157431900501, "sampling/sampling_logp_difference/max": 1.7627546787261963, "sampling/importance_sampling_ratio/min": 0.17157158255577087, "sampling/importance_sampling_ratio/mean": 1.001312494277954, "sampling/importance_sampling_ratio/max": 1.6658356189727783, "entropy": 0.07044978253543377, "clip_ratio/low_mean": 0.0031779661076143384, "clip_ratio/low_min": 0.0031779661076143384, "clip_ratio/high_mean": 0.01176445058081299, "clip_ratio/high_max": 0.01176445058081299, "clip_ratio/region_mean": 0.014942416688427329, "reward_total_mean": 0.6418830156326294, "reward_meter_mean": 0.9984818696975708, "reward_meter_std": 8.360291394637898e-05, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8035714030265808, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.6418830156326294, "reward_total_composite_std": 0.05907980352640152} {"timestamp_utc": "2026-04-12T00:10:04Z", "mode": "train", "global_step": 1421, "epoch": 0.05707514961641965, "loss": -0.006, "grad_norm": 11.100446701049805, "learning_rate": 5.696969696969698e-06, "num_tokens": 3197831.0, "completions/mean_length": 42.75, "completions/min_length": 39.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.75, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.7552404403686523, "rewards/meter/std": 0.31532391905784607, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7552404403686523, "rewards/total_composite/std": 0.31532391905784607, "reward": 0.7552404403686523, "reward_std": 0.31532391905784607, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06012430414557457, "sampling/sampling_logp_difference/max": 1.009714961051941, "sampling/importance_sampling_ratio/min": 0.36432281136512756, "sampling/importance_sampling_ratio/mean": 1.007340669631958, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3276716824620962, "clip_ratio/low_mean": 0.018345543881878257, "clip_ratio/low_min": 0.018345543881878257, "clip_ratio/high_mean": 0.049062950536608696, "clip_ratio/high_max": 0.049062950536608696, "clip_ratio/region_mean": 0.06740849441848695, "reward_total_mean": 0.7552404403686523, "reward_meter_mean": 0.7552404403686523, "reward_meter_std": 0.31532391905784607, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7552404403686523, "reward_total_composite_std": 0.31532391905784607} {"timestamp_utc": "2026-04-12T00:10:09Z", "mode": "train", "global_step": 1422, "epoch": 0.057115315098204604, "loss": 0.0018, "grad_norm": 3.4191842079162598, "learning_rate": 5.693939393939394e-06, "num_tokens": 3199274.0, "completions/mean_length": 29.375, "completions/min_length": 29.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.375, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.9963138103485107, "rewards/meter/std": 0.00021444838785100728, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9963138103485107, "rewards/total_composite/std": 0.00021444838785100728, "reward": 0.9963138103485107, "reward_std": 0.00021445513993967324, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013728239573538303, "sampling/sampling_logp_difference/max": 0.6362069845199585, "sampling/importance_sampling_ratio/min": 0.5292962193489075, "sampling/importance_sampling_ratio/mean": 1.001716136932373, "sampling/importance_sampling_ratio/max": 1.1206741333007812, "entropy": 0.09071109816431999, "clip_ratio/low_mean": 0.004310344811528921, "clip_ratio/low_min": 0.004310344811528921, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.004310344811528921, "reward_total_mean": 0.9963138103485107, "reward_meter_mean": 0.9963138103485107, "reward_meter_std": 0.00021444838785100728, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9963138103485107, "reward_total_composite_std": 0.00021444838785100728} {"timestamp_utc": "2026-04-12T00:10:13Z", "mode": "train", "global_step": 1423, "epoch": 0.05715548057998956, "loss": 0.0189, "grad_norm": 12.835563659667969, "learning_rate": 5.690909090909091e-06, "num_tokens": 3200795.0, "completions/mean_length": 34.125, "completions/min_length": 33.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.125, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9920762777328491, "rewards/meter/std": 0.0020317891612648964, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9920762777328491, "rewards/total_composite/std": 0.0020317891612648964, "reward": 0.9920762777328491, "reward_std": 0.0020317891612648964, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02493356168270111, "sampling/sampling_logp_difference/max": 0.7522921562194824, "sampling/importance_sampling_ratio/min": 0.47128504514694214, "sampling/importance_sampling_ratio/mean": 0.9959531426429749, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10061774123460054, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.014817290706560016, "clip_ratio/high_max": 0.014817290706560016, "clip_ratio/region_mean": 0.014817290706560016, "reward_total_mean": 0.9920762777328491, "reward_meter_mean": 0.9920762777328491, "reward_meter_std": 0.0020317891612648964, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9920762777328491, "reward_total_composite_std": 0.0020317891612648964} {"timestamp_utc": "2026-04-12T00:10:18Z", "mode": "train", "global_step": 1424, "epoch": 0.05719564606177451, "loss": -0.0099, "grad_norm": 5.431046485900879, "learning_rate": 5.687878787878789e-06, "num_tokens": 3202519.0, "completions/mean_length": 55.5, "completions/min_length": 53.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.5, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9895903468132019, "rewards/meter/std": 0.012036402709782124, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9895903468132019, "rewards/total_composite/std": 0.012036402709782124, "reward": 0.9895903468132019, "reward_std": 0.012036396190524101, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.044165343046188354, "sampling/sampling_logp_difference/max": 1.3421528339385986, "sampling/importance_sampling_ratio/min": 0.2612825632095337, "sampling/importance_sampling_ratio/mean": 1.0014959573745728, "sampling/importance_sampling_ratio/max": 1.5429896116256714, "entropy": 0.2528437674045563, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0493996306322515, "clip_ratio/high_max": 0.0493996306322515, "clip_ratio/region_mean": 0.0493996306322515, "reward_total_mean": 0.9895903468132019, "reward_meter_mean": 0.9895903468132019, "reward_meter_std": 0.012036402709782124, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9895903468132019, "reward_total_composite_std": 0.012036402709782124} {"timestamp_utc": "2026-04-12T00:10:26Z", "mode": "train", "global_step": 1425, "epoch": 0.057235811543559466, "loss": 0.0056, "grad_norm": 2.6007766723632812, "learning_rate": 5.684848484848485e-06, "num_tokens": 3206882.0, "completions/mean_length": 320.375, "completions/min_length": 305.0, "completions/max_length": 331.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 320.375, "completions/min_terminated_length": 305.0, "completions/max_terminated_length": 331.0, "rewards/meter/mean": 0.942986011505127, "rewards/meter/std": 0.145384281873703, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.033064987510442734, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8428308963775635, "rewards/repeat_penalty/std": 0.10538350045681, "rewards/total_composite/mean": 0.4912066161632538, "rewards/total_composite/std": 0.07578601688146591, "reward": 0.4912066161632538, "reward_std": 0.07578601688146591, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03792227804660797, "sampling/sampling_logp_difference/max": 2.1388535499572754, "sampling/importance_sampling_ratio/min": 0.11778980493545532, "sampling/importance_sampling_ratio/mean": 1.0055261850357056, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2617227640002966, "clip_ratio/low_mean": 0.006616426864638925, "clip_ratio/low_min": 0.006616426864638925, "clip_ratio/high_mean": 0.018805116647854447, "clip_ratio/high_max": 0.018805116647854447, "clip_ratio/region_mean": 0.025421543512493372, "reward_total_mean": 0.4912066161632538, "reward_meter_mean": 0.942986011505127, "reward_meter_std": 0.145384281873703, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.033064987510442734, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8428308963775635, "reward_repeat_penalty_std": 0.10538350045681, "reward_total_composite_mean": 0.4912066161632538, "reward_total_composite_std": 0.07578601688146591} {"timestamp_utc": "2026-04-12T00:10:31Z", "mode": "train", "global_step": 1426, "epoch": 0.05727597702534442, "loss": -0.01, "grad_norm": 12.311224937438965, "learning_rate": 5.681818181818183e-06, "num_tokens": 3208730.0, "completions/mean_length": 63.0, "completions/min_length": 60.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.0, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.996849000453949, "rewards/meter/std": 0.0023078599479049444, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.996849000453949, "rewards/total_composite/std": 0.0023078599479049444, "reward": 0.996849000453949, "reward_std": 0.002307864371687174, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03911422938108444, "sampling/sampling_logp_difference/max": 1.8360328674316406, "sampling/importance_sampling_ratio/min": 0.15944872796535492, "sampling/importance_sampling_ratio/mean": 1.0055710077285767, "sampling/importance_sampling_ratio/max": 1.7190132141113281, "entropy": 0.2404075190424919, "clip_ratio/low_mean": 0.006051587639376521, "clip_ratio/low_min": 0.006051587639376521, "clip_ratio/high_mean": 0.043533024145290256, "clip_ratio/high_max": 0.043533024145290256, "clip_ratio/region_mean": 0.04958461178466678, "reward_total_mean": 0.996849000453949, "reward_meter_mean": 0.996849000453949, "reward_meter_std": 0.0023078599479049444, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.996849000453949, "reward_total_composite_std": 0.0023078599479049444} {"timestamp_utc": "2026-04-12T00:10:36Z", "mode": "train", "global_step": 1427, "epoch": 0.057316142507129374, "loss": -0.0018, "grad_norm": 2.7146718502044678, "learning_rate": 5.67878787878788e-06, "num_tokens": 3210997.0, "completions/mean_length": 90.375, "completions/min_length": 89.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.375, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.9984976053237915, "rewards/meter/std": 9.82403289526701e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.9485726356506348, "rewards/total_composite/std": 0.09244248270988464, "reward": 0.9485726356506348, "reward_std": 0.09244248270988464, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014895378611981869, "sampling/sampling_logp_difference/max": 1.0964069366455078, "sampling/importance_sampling_ratio/min": 0.33406925201416016, "sampling/importance_sampling_ratio/mean": 0.9975816011428833, "sampling/importance_sampling_ratio/max": 1.3982148170471191, "entropy": 0.06489505246281624, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.012471178779378533, "clip_ratio/high_max": 0.012471178779378533, "clip_ratio/region_mean": 0.012471178779378533, "reward_total_mean": 0.9485726356506348, "reward_meter_mean": 0.9984976053237915, "reward_meter_std": 9.82403289526701e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.9485726356506348, "reward_total_composite_std": 0.09244248270988464} {"timestamp_utc": "2026-04-12T00:10:43Z", "mode": "train", "global_step": 1428, "epoch": 0.05735630798891433, "loss": 0.0093, "grad_norm": 2.3229403495788574, "learning_rate": 5.675757575757577e-06, "num_tokens": 3214669.0, "completions/mean_length": 256.0, "completions/min_length": 245.0, "completions/max_length": 272.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 256.0, "completions/min_terminated_length": 245.0, "completions/max_terminated_length": 272.0, "rewards/meter/mean": 0.9840718507766724, "rewards/meter/std": 0.03672640025615692, "rewards/count_adherence/mean": 0.699999988079071, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8125, "rewards/repeat_penalty/std": 0.10492872446775436, "rewards/total_composite/mean": 0.5593699216842651, "rewards/total_composite/std": 0.07350600510835648, "reward": 0.5593699216842651, "reward_std": 0.07350601255893707, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0390043631196022, "sampling/sampling_logp_difference/max": 3.3374457359313965, "sampling/importance_sampling_ratio/min": 0.035527586936950684, "sampling/importance_sampling_ratio/mean": 1.0045963525772095, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.24354491382837296, "clip_ratio/low_mean": 0.01068126189056784, "clip_ratio/low_min": 0.01068126189056784, "clip_ratio/high_mean": 0.012853078544139862, "clip_ratio/high_max": 0.012853078544139862, "clip_ratio/region_mean": 0.0235343404347077, "reward_total_mean": 0.5593699216842651, "reward_meter_mean": 0.9840718507766724, "reward_meter_std": 0.03672640025615692, "reward_count_adherence_mean": 0.699999988079071, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8125, "reward_repeat_penalty_std": 0.10492872446775436, "reward_total_composite_mean": 0.5593699216842651, "reward_total_composite_std": 0.07350600510835648} {"timestamp_utc": "2026-04-12T00:10:48Z", "mode": "train", "global_step": 1429, "epoch": 0.05739647347069928, "loss": 0.0002, "grad_norm": 2.9396092891693115, "learning_rate": 5.672727272727273e-06, "num_tokens": 3217095.0, "completions/mean_length": 123.25, "completions/min_length": 122.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 123.25, "completions/min_terminated_length": 122.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.9978071451187134, "rewards/meter/std": 0.0011365336831659079, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8035714030265808, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.8018503189086914, "rewards/total_composite/std": 0.07426349073648453, "reward": 0.8018503189086914, "reward_std": 0.07426347583532333, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015226030722260475, "sampling/sampling_logp_difference/max": 1.5297276973724365, "sampling/importance_sampling_ratio/min": 0.2165946513414383, "sampling/importance_sampling_ratio/mean": 0.999223530292511, "sampling/importance_sampling_ratio/max": 1.302078366279602, "entropy": 0.07903782464563847, "clip_ratio/low_mean": 0.006106290849857032, "clip_ratio/low_min": 0.006106290849857032, "clip_ratio/high_mean": 0.013211918994784355, "clip_ratio/high_max": 0.013211918994784355, "clip_ratio/region_mean": 0.019318209844641387, "reward_total_mean": 0.8018503189086914, "reward_meter_mean": 0.9978071451187134, "reward_meter_std": 0.0011365336831659079, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8035714030265808, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.8018503189086914, "reward_total_composite_std": 0.07426349073648453} {"timestamp_utc": "2026-04-12T00:10:53Z", "mode": "train", "global_step": 1430, "epoch": 0.057436638952484236, "loss": 0.0069, "grad_norm": 5.2827301025390625, "learning_rate": 5.6696969696969705e-06, "num_tokens": 3218659.0, "completions/mean_length": 56.5, "completions/min_length": 53.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.5, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.43106603622436523, "rewards/meter/std": 0.3671012222766876, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.43106603622436523, "rewards/total_composite/std": 0.3671012222766876, "reward": 0.43106603622436523, "reward_std": 0.36710119247436523, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.059371624141931534, "sampling/sampling_logp_difference/max": 1.259765625, "sampling/importance_sampling_ratio/min": 0.2837205231189728, "sampling/importance_sampling_ratio/mean": 1.0158692598342896, "sampling/importance_sampling_ratio/max": 1.4617520570755005, "entropy": 0.5998508296906948, "clip_ratio/low_mean": 0.030917756259441376, "clip_ratio/low_min": 0.030917756259441376, "clip_ratio/high_mean": 0.008850250858813524, "clip_ratio/high_max": 0.008850250858813524, "clip_ratio/region_mean": 0.0397680071182549, "reward_total_mean": 0.43106603622436523, "reward_meter_mean": 0.43106603622436523, "reward_meter_std": 0.3671012222766876, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.43106603622436523, "reward_total_composite_std": 0.3671012222766876} {"timestamp_utc": "2026-04-12T00:10:58Z", "mode": "train", "global_step": 1431, "epoch": 0.05747680443426919, "loss": -0.0112, "grad_norm": 3.8664066791534424, "learning_rate": 5.666666666666667e-06, "num_tokens": 3220585.0, "completions/mean_length": 90.75, "completions/min_length": 87.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.75, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.9018567800521851, "rewards/meter/std": 0.26490822434425354, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9018567800521851, "rewards/total_composite/std": 0.26490822434425354, "reward": 0.9018567800521851, "reward_std": 0.2649082541465759, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.042072996497154236, "sampling/sampling_logp_difference/max": 1.666844367980957, "sampling/importance_sampling_ratio/min": 0.1888420283794403, "sampling/importance_sampling_ratio/mean": 1.0014922618865967, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12930846586823463, "clip_ratio/low_mean": 0.008522727526724339, "clip_ratio/low_min": 0.008522727526724339, "clip_ratio/high_mean": 0.027468874352052808, "clip_ratio/high_max": 0.027468874352052808, "clip_ratio/region_mean": 0.035991601878777146, "reward_total_mean": 0.9018567800521851, "reward_meter_mean": 0.9018567800521851, "reward_meter_std": 0.26490822434425354, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9018567800521851, "reward_total_composite_std": 0.26490822434425354} {"timestamp_utc": "2026-04-12T00:11:03Z", "mode": "train", "global_step": 1432, "epoch": 0.05751696991605414, "loss": 0.002, "grad_norm": 3.0692851543426514, "learning_rate": 5.663636363636364e-06, "num_tokens": 3222928.0, "completions/mean_length": 118.875, "completions/min_length": 118.0, "completions/max_length": 120.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.875, "completions/min_terminated_length": 118.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.9983639717102051, "rewards/meter/std": 0.00039870149339549243, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.11921755969524384, "rewards/total_composite/mean": 0.7309507131576538, "rewards/total_composite/std": 0.11907364428043365, "reward": 0.7309507131576538, "reward_std": 0.11907363682985306, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022948887199163437, "sampling/sampling_logp_difference/max": 1.848806619644165, "sampling/importance_sampling_ratio/min": 0.1574249118566513, "sampling/importance_sampling_ratio/mean": 1.0022363662719727, "sampling/importance_sampling_ratio/max": 1.6000510454177856, "entropy": 0.1475573442876339, "clip_ratio/low_mean": 0.0031601624796167016, "clip_ratio/low_min": 0.0031601624796167016, "clip_ratio/high_mean": 0.007326977560296655, "clip_ratio/high_max": 0.007326977560296655, "clip_ratio/region_mean": 0.010487140039913356, "reward_total_mean": 0.7309507131576538, "reward_meter_mean": 0.9983639717102051, "reward_meter_std": 0.00039870149339549243, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.11921755969524384, "reward_total_composite_mean": 0.7309507131576538, "reward_total_composite_std": 0.11907364428043365} {"timestamp_utc": "2026-04-12T00:11:08Z", "mode": "train", "global_step": 1433, "epoch": 0.0575571353978391, "loss": -0.0086, "grad_norm": 4.21342658996582, "learning_rate": 5.6606060606060606e-06, "num_tokens": 3224674.0, "completions/mean_length": 64.25, "completions/min_length": 60.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.25, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.7867549657821655, "rewards/meter/std": 0.2720617949962616, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7867549657821655, "rewards/total_composite/std": 0.2720617949962616, "reward": 0.7867549657821655, "reward_std": 0.2720617651939392, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.045874159783124924, "sampling/sampling_logp_difference/max": 1.6135480403900146, "sampling/importance_sampling_ratio/min": 0.19917964935302734, "sampling/importance_sampling_ratio/mean": 1.0058132410049438, "sampling/importance_sampling_ratio/max": 1.7156156301498413, "entropy": 0.3227595426142216, "clip_ratio/low_mean": 0.01948191737756133, "clip_ratio/low_min": 0.01948191737756133, "clip_ratio/high_mean": 0.017400611890479922, "clip_ratio/high_max": 0.017400611890479922, "clip_ratio/region_mean": 0.03688252926804125, "reward_total_mean": 0.7867549657821655, "reward_meter_mean": 0.7867549657821655, "reward_meter_std": 0.2720617949962616, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7867549657821655, "reward_total_composite_std": 0.2720617949962616} {"timestamp_utc": "2026-04-12T00:11:12Z", "mode": "train", "global_step": 1434, "epoch": 0.05759730087962405, "loss": -0.0555, "grad_norm": 7.206746578216553, "learning_rate": 5.657575757575759e-06, "num_tokens": 3226467.0, "completions/mean_length": 63.125, "completions/min_length": 54.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.125, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9712523221969604, "rewards/meter/std": 0.06995871663093567, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9712523221969604, "rewards/total_composite/std": 0.06995871663093567, "reward": 0.9712523221969604, "reward_std": 0.06995872408151627, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024187438189983368, "sampling/sampling_logp_difference/max": 1.3712058067321777, "sampling/importance_sampling_ratio/min": 0.25380071997642517, "sampling/importance_sampling_ratio/mean": 1.0004560947418213, "sampling/importance_sampling_ratio/max": 1.8253268003463745, "entropy": 0.12332060746848583, "clip_ratio/low_mean": 0.002314814832061529, "clip_ratio/low_min": 0.002314814832061529, "clip_ratio/high_mean": 0.0194749265210703, "clip_ratio/high_max": 0.0194749265210703, "clip_ratio/region_mean": 0.02178974135313183, "reward_total_mean": 0.9712523221969604, "reward_meter_mean": 0.9712523221969604, "reward_meter_std": 0.06995871663093567, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9712523221969604, "reward_total_composite_std": 0.06995871663093567} {"timestamp_utc": "2026-04-12T00:11:17Z", "mode": "train", "global_step": 1435, "epoch": 0.057637466361409005, "loss": -0.0, "grad_norm": 6.715077877044678, "learning_rate": 5.654545454545455e-06, "num_tokens": 3228023.0, "completions/mean_length": 45.5, "completions/min_length": 43.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.5, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.8577103614807129, "rewards/meter/std": 0.20525795221328735, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8577103614807129, "rewards/total_composite/std": 0.20525795221328735, "reward": 0.8577103614807129, "reward_std": 0.20525793731212616, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021754316985607147, "sampling/sampling_logp_difference/max": 0.6161007881164551, "sampling/importance_sampling_ratio/min": 0.5400460958480835, "sampling/importance_sampling_ratio/mean": 0.9992294311523438, "sampling/importance_sampling_ratio/max": 1.5408525466918945, "entropy": 0.1447599595412612, "clip_ratio/low_mean": 0.00872093066573143, "clip_ratio/low_min": 0.00872093066573143, "clip_ratio/high_mean": 0.019032032461836934, "clip_ratio/high_max": 0.019032032461836934, "clip_ratio/region_mean": 0.027752963127568364, "reward_total_mean": 0.8577103614807129, "reward_meter_mean": 0.8577103614807129, "reward_meter_std": 0.20525795221328735, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8577103614807129, "reward_total_composite_std": 0.20525795221328735} {"timestamp_utc": "2026-04-12T00:11:22Z", "mode": "train", "global_step": 1436, "epoch": 0.05767763184319396, "loss": -0.0268, "grad_norm": 5.0643205642700195, "learning_rate": 5.651515151515152e-06, "num_tokens": 3229879.0, "completions/mean_length": 64.0, "completions/min_length": 60.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9666499495506287, "rewards/meter/std": 0.03790951892733574, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9666499495506287, "rewards/total_composite/std": 0.03790951892733574, "reward": 0.9666499495506287, "reward_std": 0.03790954127907753, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030510857701301575, "sampling/sampling_logp_difference/max": 1.486882209777832, "sampling/importance_sampling_ratio/min": 0.2260764092206955, "sampling/importance_sampling_ratio/mean": 1.0077619552612305, "sampling/importance_sampling_ratio/max": 1.844899296760559, "entropy": 0.13815412484109402, "clip_ratio/low_mean": 0.008230874547734857, "clip_ratio/low_min": 0.008230874547734857, "clip_ratio/high_mean": 0.024871644098311663, "clip_ratio/high_max": 0.024871644098311663, "clip_ratio/region_mean": 0.03310251864604652, "reward_total_mean": 0.9666499495506287, "reward_meter_mean": 0.9666499495506287, "reward_meter_std": 0.03790951892733574, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9666499495506287, "reward_total_composite_std": 0.03790951892733574} {"timestamp_utc": "2026-04-12T00:11:26Z", "mode": "train", "global_step": 1437, "epoch": 0.05771779732497891, "loss": -0.0032, "grad_norm": 7.082005500793457, "learning_rate": 5.648484848484849e-06, "num_tokens": 3231833.0, "completions/mean_length": 64.25, "completions/min_length": 61.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.25, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.990328311920166, "rewards/meter/std": 0.005682698916643858, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.990328311920166, "rewards/total_composite/std": 0.005682698916643858, "reward": 0.990328311920166, "reward_std": 0.005682693794369698, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0303493719547987, "sampling/sampling_logp_difference/max": 2.149282455444336, "sampling/importance_sampling_ratio/min": 0.11656776815652847, "sampling/importance_sampling_ratio/mean": 0.9941160082817078, "sampling/importance_sampling_ratio/max": 1.7731237411499023, "entropy": 0.1368792960420251, "clip_ratio/low_mean": 0.005988386110402644, "clip_ratio/low_min": 0.005988386110402644, "clip_ratio/high_mean": 0.005740093300119042, "clip_ratio/high_max": 0.005740093300119042, "clip_ratio/region_mean": 0.011728479410521686, "reward_total_mean": 0.990328311920166, "reward_meter_mean": 0.990328311920166, "reward_meter_std": 0.005682698916643858, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.990328311920166, "reward_total_composite_std": 0.005682698916643858} {"timestamp_utc": "2026-04-12T00:11:31Z", "mode": "train", "global_step": 1438, "epoch": 0.05775796280676387, "loss": -0.009, "grad_norm": 4.488318920135498, "learning_rate": 5.645454545454546e-06, "num_tokens": 3233673.0, "completions/mean_length": 54.0, "completions/min_length": 47.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9756955504417419, "rewards/meter/std": 0.017869651317596436, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9756955504417419, "rewards/total_composite/std": 0.017869651317596436, "reward": 0.9756955504417419, "reward_std": 0.01786964386701584, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03772404417395592, "sampling/sampling_logp_difference/max": 1.3473641872406006, "sampling/importance_sampling_ratio/min": 0.2599244713783264, "sampling/importance_sampling_ratio/mean": 1.008415699005127, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.21246233023703098, "clip_ratio/low_mean": 0.007591875968500972, "clip_ratio/low_min": 0.007591875968500972, "clip_ratio/high_mean": 0.018068846315145493, "clip_ratio/high_max": 0.018068846315145493, "clip_ratio/region_mean": 0.025660722283646464, "reward_total_mean": 0.9756955504417419, "reward_meter_mean": 0.9756955504417419, "reward_meter_std": 0.017869651317596436, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9756955504417419, "reward_total_composite_std": 0.017869651317596436} {"timestamp_utc": "2026-04-12T00:11:36Z", "mode": "train", "global_step": 1439, "epoch": 0.05779812828854882, "loss": -0.0046, "grad_norm": 3.8302407264709473, "learning_rate": 5.642424242424242e-06, "num_tokens": 3235514.0, "completions/mean_length": 66.125, "completions/min_length": 59.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.125, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9976571798324585, "rewards/meter/std": 0.001123868627473712, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9976571798324585, "rewards/total_composite/std": 0.001123868627473712, "reward": 0.9976571798324585, "reward_std": 0.001123874681070447, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04735542833805084, "sampling/sampling_logp_difference/max": 1.1300544738769531, "sampling/importance_sampling_ratio/min": 0.3230156898498535, "sampling/importance_sampling_ratio/mean": 1.0110960006713867, "sampling/importance_sampling_ratio/max": 1.605924129486084, "entropy": 0.2875422090291977, "clip_ratio/low_mean": 0.007575757801532745, "clip_ratio/low_min": 0.007575757801532745, "clip_ratio/high_mean": 0.03688650333788246, "clip_ratio/high_max": 0.03688650333788246, "clip_ratio/region_mean": 0.044462261139415205, "reward_total_mean": 0.9976571798324585, "reward_meter_mean": 0.9976571798324585, "reward_meter_std": 0.001123868627473712, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9976571798324585, "reward_total_composite_std": 0.001123868627473712} {"timestamp_utc": "2026-04-12T00:11:42Z", "mode": "train", "global_step": 1440, "epoch": 0.057838293770333775, "loss": -0.0197, "grad_norm": 2.74971342086792, "learning_rate": 5.6393939393939405e-06, "num_tokens": 3238860.0, "completions/mean_length": 197.25, "completions/min_length": 188.0, "completions/max_length": 211.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 197.25, "completions/min_terminated_length": 188.0, "completions/max_terminated_length": 211.0, "rewards/meter/mean": 0.9103585481643677, "rewards/meter/std": 0.11414875090122223, "rewards/count_adherence/mean": 0.8392857313156128, "rewards/count_adherence/std": 0.05050762742757797, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8397727012634277, "rewards/repeat_penalty/std": 0.1051156297326088, "rewards/total_composite/mean": 0.641140341758728, "rewards/total_composite/std": 0.11881579458713531, "reward": 0.641140341758728, "reward_std": 0.11881577223539352, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.052030667662620544, "sampling/sampling_logp_difference/max": 1.4830026626586914, "sampling/importance_sampling_ratio/min": 0.22695519030094147, "sampling/importance_sampling_ratio/mean": 1.0104938745498657, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3795258644968271, "clip_ratio/low_mean": 0.01373849913943559, "clip_ratio/low_min": 0.01373849913943559, "clip_ratio/high_mean": 0.025198507588356733, "clip_ratio/high_max": 0.025198507588356733, "clip_ratio/region_mean": 0.03893700672779232, "reward_total_mean": 0.641140341758728, "reward_meter_mean": 0.9103585481643677, "reward_meter_std": 0.11414875090122223, "reward_count_adherence_mean": 0.8392857313156128, "reward_count_adherence_std": 0.05050762742757797, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8397727012634277, "reward_repeat_penalty_std": 0.1051156297326088, "reward_total_composite_mean": 0.641140341758728, "reward_total_composite_std": 0.11881579458713531} {"timestamp_utc": "2026-04-12T00:11:46Z", "mode": "train", "global_step": 1441, "epoch": 0.05787845925211873, "loss": -0.0023, "grad_norm": 5.571752548217773, "learning_rate": 5.636363636363636e-06, "num_tokens": 3240498.0, "completions/mean_length": 54.75, "completions/min_length": 54.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.75, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.9891074895858765, "rewards/meter/std": 0.0014680020976811647, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9891074895858765, "rewards/total_composite/std": 0.0014680020976811647, "reward": 0.9891074895858765, "reward_std": 0.0014680068707093596, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022303204983472824, "sampling/sampling_logp_difference/max": 0.8718528747558594, "sampling/importance_sampling_ratio/min": 0.41817599534988403, "sampling/importance_sampling_ratio/mean": 1.0070688724517822, "sampling/importance_sampling_ratio/max": 1.8307318687438965, "entropy": 0.15620220359414816, "clip_ratio/low_mean": 0.009132996667176485, "clip_ratio/low_min": 0.009132996667176485, "clip_ratio/high_mean": 0.004545454401522875, "clip_ratio/high_max": 0.004545454401522875, "clip_ratio/region_mean": 0.01367845106869936, "reward_total_mean": 0.9891074895858765, "reward_meter_mean": 0.9891074895858765, "reward_meter_std": 0.0014680020976811647, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9891074895858765, "reward_total_composite_std": 0.0014680020976811647} {"timestamp_utc": "2026-04-12T00:11:52Z", "mode": "train", "global_step": 1442, "epoch": 0.05791862473390368, "loss": 0.0076, "grad_norm": 4.191330909729004, "learning_rate": 5.633333333333334e-06, "num_tokens": 3243758.0, "completions/mean_length": 168.5, "completions/min_length": 145.0, "completions/max_length": 182.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 168.5, "completions/min_terminated_length": 145.0, "completions/max_terminated_length": 182.0, "rewards/meter/mean": 0.12805184721946716, "rewards/meter/std": 0.1377500742673874, "rewards/count_adherence/mean": 0.6250000596046448, "rewards/count_adherence/std": 0.05750546231865883, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9217171669006348, "rewards/repeat_penalty/std": 0.1535457819700241, "rewards/total_composite/mean": 0.07739339768886566, "rewards/total_composite/std": 0.08249951899051666, "reward": 0.07739339768886566, "reward_std": 0.08249951899051666, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.089544378221035, "sampling/sampling_logp_difference/max": 3.6343159675598145, "sampling/importance_sampling_ratio/min": 0.02640198916196823, "sampling/importance_sampling_ratio/mean": 1.0114877223968506, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7288497872650623, "clip_ratio/low_mean": 0.0417820424772799, "clip_ratio/low_min": 0.0417820424772799, "clip_ratio/high_mean": 0.030609328765422106, "clip_ratio/high_max": 0.030609328765422106, "clip_ratio/region_mean": 0.07239137124270201, "reward_total_mean": 0.07739339768886566, "reward_meter_mean": 0.12805184721946716, "reward_meter_std": 0.1377500742673874, "reward_count_adherence_mean": 0.6250000596046448, "reward_count_adherence_std": 0.05750546231865883, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9217171669006348, "reward_repeat_penalty_std": 0.1535457819700241, "reward_total_composite_mean": 0.07739339768886566, "reward_total_composite_std": 0.08249951899051666} {"timestamp_utc": "2026-04-12T00:12:00Z", "mode": "train", "global_step": 1443, "epoch": 0.05795879021568864, "loss": -0.0031, "grad_norm": 2.3659539222717285, "learning_rate": 5.630303030303031e-06, "num_tokens": 3247759.0, "completions/mean_length": 274.125, "completions/min_length": 263.0, "completions/max_length": 293.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 274.125, "completions/min_terminated_length": 263.0, "completions/max_terminated_length": 293.0, "rewards/meter/mean": 0.993506908416748, "rewards/meter/std": 0.008024541661143303, "rewards/count_adherence/mean": 0.7250000238418579, "rewards/count_adherence/std": 0.04629101604223251, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8730769157409668, "rewards/repeat_penalty/std": 0.05622680485248566, "rewards/total_composite/mean": 0.6270208954811096, "rewards/total_composite/std": 0.023548215627670288, "reward": 0.6270208954811096, "reward_std": 0.023548215627670288, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04566960409283638, "sampling/sampling_logp_difference/max": 2.6650161743164062, "sampling/importance_sampling_ratio/min": 0.0695982277393341, "sampling/importance_sampling_ratio/mean": 1.0080655813217163, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.34477731212973595, "clip_ratio/low_mean": 0.008868969744071364, "clip_ratio/low_min": 0.008868969744071364, "clip_ratio/high_mean": 0.026962524512782693, "clip_ratio/high_max": 0.026962524512782693, "clip_ratio/region_mean": 0.03583149425685406, "reward_total_mean": 0.6270208954811096, "reward_meter_mean": 0.993506908416748, "reward_meter_std": 0.008024541661143303, "reward_count_adherence_mean": 0.7250000238418579, "reward_count_adherence_std": 0.04629101604223251, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8730769157409668, "reward_repeat_penalty_std": 0.05622680485248566, "reward_total_composite_mean": 0.6270208954811096, "reward_total_composite_std": 0.023548215627670288} {"timestamp_utc": "2026-04-12T00:12:05Z", "mode": "train", "global_step": 1444, "epoch": 0.05799895569747359, "loss": -0.0009, "grad_norm": 3.132058620452881, "learning_rate": 5.627272727272728e-06, "num_tokens": 3250457.0, "completions/mean_length": 150.25, "completions/min_length": 143.0, "completions/max_length": 154.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 150.25, "completions/min_terminated_length": 143.0, "completions/max_terminated_length": 154.0, "rewards/meter/mean": 0.9959889054298401, "rewards/meter/std": 0.00317049166187644, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.9605348110198975, "rewards/total_composite/std": 0.06770440191030502, "reward": 0.9605348110198975, "reward_std": 0.06770440936088562, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05538027361035347, "sampling/sampling_logp_difference/max": 1.4519987106323242, "sampling/importance_sampling_ratio/min": 0.23410192131996155, "sampling/importance_sampling_ratio/mean": 1.0069429874420166, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3815644532442093, "clip_ratio/low_mean": 0.006868131808005273, "clip_ratio/low_min": 0.006868131808005273, "clip_ratio/high_mean": 0.035727029433473945, "clip_ratio/high_max": 0.035727029433473945, "clip_ratio/region_mean": 0.04259516124147922, "reward_total_mean": 0.9605348110198975, "reward_meter_mean": 0.9959889054298401, "reward_meter_std": 0.00317049166187644, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.9605348110198975, "reward_total_composite_std": 0.06770440191030502} {"timestamp_utc": "2026-04-12T00:12:12Z", "mode": "train", "global_step": 1445, "epoch": 0.058039121179258545, "loss": 0.0052, "grad_norm": 3.3944966793060303, "learning_rate": 5.624242424242424e-06, "num_tokens": 3253588.0, "completions/mean_length": 188.375, "completions/min_length": 182.0, "completions/max_length": 196.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 188.375, "completions/min_terminated_length": 182.0, "completions/max_terminated_length": 196.0, "rewards/meter/mean": 0.9968078136444092, "rewards/meter/std": 0.0010806269710883498, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9444444179534912, "rewards/repeat_penalty/std": 0.08399210125207901, "rewards/total_composite/mean": 0.9414394497871399, "rewards/total_composite/std": 0.08388549834489822, "reward": 0.9414394497871399, "reward_std": 0.08388549834489822, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.050331130623817444, "sampling/sampling_logp_difference/max": 1.6728553771972656, "sampling/importance_sampling_ratio/min": 0.18771031498908997, "sampling/importance_sampling_ratio/mean": 1.013535737991333, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.38519781827926636, "clip_ratio/low_mean": 0.012751690810546279, "clip_ratio/low_min": 0.012751690810546279, "clip_ratio/high_mean": 0.02173018571920693, "clip_ratio/high_max": 0.02173018571920693, "clip_ratio/region_mean": 0.03448187652975321, "reward_total_mean": 0.9414394497871399, "reward_meter_mean": 0.9968078136444092, "reward_meter_std": 0.0010806269710883498, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9444444179534912, "reward_repeat_penalty_std": 0.08399210125207901, "reward_total_composite_mean": 0.9414394497871399, "reward_total_composite_std": 0.08388549834489822} {"timestamp_utc": "2026-04-12T00:12:16Z", "mode": "train", "global_step": 1446, "epoch": 0.0580792866610435, "loss": 0.0122, "grad_norm": 5.573071479797363, "learning_rate": 5.6212121212121215e-06, "num_tokens": 3255370.0, "completions/mean_length": 70.75, "completions/min_length": 69.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.75, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.994280993938446, "rewards/meter/std": 0.0039334069006145, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.994280993938446, "rewards/total_composite/std": 0.0039334069006145, "reward": 0.994280993938446, "reward_std": 0.003933423198759556, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07322768121957779, "sampling/sampling_logp_difference/max": 1.5738625526428223, "sampling/importance_sampling_ratio/min": 0.2072431445121765, "sampling/importance_sampling_ratio/mean": 1.0078636407852173, "sampling/importance_sampling_ratio/max": 1.8624515533447266, "entropy": 0.48965371400117874, "clip_ratio/low_mean": 0.017605633474886417, "clip_ratio/low_min": 0.017605633474886417, "clip_ratio/high_mean": 0.05117733031511307, "clip_ratio/high_max": 0.05117733031511307, "clip_ratio/region_mean": 0.06878296378999949, "reward_total_mean": 0.994280993938446, "reward_meter_mean": 0.994280993938446, "reward_meter_std": 0.0039334069006145, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.994280993938446, "reward_total_composite_std": 0.0039334069006145} {"timestamp_utc": "2026-04-12T00:12:21Z", "mode": "train", "global_step": 1447, "epoch": 0.05811945214282845, "loss": 0.0006, "grad_norm": 7.160624980926514, "learning_rate": 5.618181818181818e-06, "num_tokens": 3256876.0, "completions/mean_length": 35.25, "completions/min_length": 34.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.25, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.9780558943748474, "rewards/meter/std": 0.022425083443522453, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9780558943748474, "rewards/total_composite/std": 0.022425083443522453, "reward": 0.9780558943748474, "reward_std": 0.022425075992941856, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04081781581044197, "sampling/sampling_logp_difference/max": 0.8824992179870605, "sampling/importance_sampling_ratio/min": 0.4137475788593292, "sampling/importance_sampling_ratio/mean": 1.0069624185562134, "sampling/importance_sampling_ratio/max": 1.6601144075393677, "entropy": 0.27801981568336487, "clip_ratio/low_mean": 0.0173611119389534, "clip_ratio/low_min": 0.0173611119389534, "clip_ratio/high_mean": 0.03094316739588976, "clip_ratio/high_max": 0.03094316739588976, "clip_ratio/region_mean": 0.04830427933484316, "reward_total_mean": 0.9780558943748474, "reward_meter_mean": 0.9780558943748474, "reward_meter_std": 0.022425083443522453, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9780558943748474, "reward_total_composite_std": 0.022425083443522453} {"timestamp_utc": "2026-04-12T00:12:26Z", "mode": "train", "global_step": 1448, "epoch": 0.058159617624613406, "loss": 0.0162, "grad_norm": 2.8199713230133057, "learning_rate": 5.615151515151516e-06, "num_tokens": 3259212.0, "completions/mean_length": 114.0, "completions/min_length": 111.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.0, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.9958944320678711, "rewards/meter/std": 0.0037708207964897156, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9709444046020508, "rewards/total_composite/std": 0.06981504708528519, "reward": 0.9709444046020508, "reward_std": 0.06981503963470459, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05924813821911812, "sampling/sampling_logp_difference/max": 1.7622275352478027, "sampling/importance_sampling_ratio/min": 0.17166204750537872, "sampling/importance_sampling_ratio/mean": 1.0023696422576904, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.348561218008399, "clip_ratio/low_mean": 0.0032051282469183207, "clip_ratio/low_min": 0.0032051282469183207, "clip_ratio/high_mean": 0.04604411777108908, "clip_ratio/high_max": 0.04604411777108908, "clip_ratio/region_mean": 0.0492492460180074, "reward_total_mean": 0.9709444046020508, "reward_meter_mean": 0.9958944320678711, "reward_meter_std": 0.0037708207964897156, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9709444046020508, "reward_total_composite_std": 0.06981504708528519} {"timestamp_utc": "2026-04-12T00:12:32Z", "mode": "train", "global_step": 1449, "epoch": 0.05819978310639836, "loss": 0.0176, "grad_norm": 2.3722407817840576, "learning_rate": 5.612121212121212e-06, "num_tokens": 3262076.0, "completions/mean_length": 171.0, "completions/min_length": 165.0, "completions/max_length": 178.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 171.0, "completions/min_terminated_length": 165.0, "completions/max_terminated_length": 178.0, "rewards/meter/mean": 0.4469248056411743, "rewards/meter/std": 0.35773399472236633, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7916666865348816, "rewards/repeat_penalty/std": 0.07120776921510696, "rewards/total_composite/mean": 0.29604610800743103, "rewards/total_composite/std": 0.23469020426273346, "reward": 0.29604610800743103, "reward_std": 0.23469018936157227, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025468070060014725, "sampling/sampling_logp_difference/max": 2.328211784362793, "sampling/importance_sampling_ratio/min": 0.09746988862752914, "sampling/importance_sampling_ratio/mean": 1.0035395622253418, "sampling/importance_sampling_ratio/max": 1.7698057889938354, "entropy": 0.16733690444380045, "clip_ratio/low_mean": 0.007986687181983143, "clip_ratio/low_min": 0.007986687181983143, "clip_ratio/high_mean": 0.011789240641519427, "clip_ratio/high_max": 0.011789240641519427, "clip_ratio/region_mean": 0.01977592782350257, "reward_total_mean": 0.29604610800743103, "reward_meter_mean": 0.4469248056411743, "reward_meter_std": 0.35773399472236633, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7916666865348816, "reward_repeat_penalty_std": 0.07120776921510696, "reward_total_composite_mean": 0.29604610800743103, "reward_total_composite_std": 0.23469020426273346} {"timestamp_utc": "2026-04-12T00:12:38Z", "mode": "train", "global_step": 1450, "epoch": 0.058239948588183314, "loss": 0.0088, "grad_norm": 2.1575472354888916, "learning_rate": 5.60909090909091e-06, "num_tokens": 3265160.0, "completions/mean_length": 192.5, "completions/min_length": 184.0, "completions/max_length": 197.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 192.5, "completions/min_terminated_length": 184.0, "completions/max_terminated_length": 197.0, "rewards/meter/mean": 0.9949524998664856, "rewards/meter/std": 0.0050539844669401646, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9166666269302368, "rewards/repeat_penalty/std": 0.07856741547584534, "rewards/total_composite/mean": 0.911919116973877, "rewards/total_composite/std": 0.0767272487282753, "reward": 0.911919116973877, "reward_std": 0.07672726362943649, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04098372161388397, "sampling/sampling_logp_difference/max": 2.1571097373962402, "sampling/importance_sampling_ratio/min": 0.11565891653299332, "sampling/importance_sampling_ratio/mean": 1.0060464143753052, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.28528641909360886, "clip_ratio/low_mean": 0.017584156128577888, "clip_ratio/low_min": 0.017584156128577888, "clip_ratio/high_mean": 0.015601763967424631, "clip_ratio/high_max": 0.015601763967424631, "clip_ratio/region_mean": 0.03318592009600252, "reward_total_mean": 0.911919116973877, "reward_meter_mean": 0.9949524998664856, "reward_meter_std": 0.0050539844669401646, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9166666269302368, "reward_repeat_penalty_std": 0.07856741547584534, "reward_total_composite_mean": 0.911919116973877, "reward_total_composite_std": 0.0767272487282753} {"timestamp_utc": "2026-04-12T00:13:33Z", "mode": "eval", "global_step": 1450, "epoch": 0.058239948588183314, "eval_loss": NaN, "eval_runtime": 55.2927, "eval_samples_per_second": 1.881, "eval_steps_per_second": 0.235, "eval_num_tokens": 3265160.0, "eval_completions/mean_length": 160.5, "eval_completions/min_length": 60.46153846153846, "eval_completions/max_length": 283.0769230769231, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 160.5, "eval_completions/min_terminated_length": 60.46153846153846, "eval_completions/max_terminated_length": 283.0769230769231, "eval_rewards/meter/mean": 0.5874437128122036, "eval_rewards/meter/std": 0.4289410481086144, "eval_rewards/count_adherence/mean": 0.8191676002282363, "eval_rewards/count_adherence/std": 0.17260103730055001, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/repeat_penalty/mean": 0.8410639992127051, "eval_rewards/repeat_penalty/std": 0.16562061241039863, "eval_rewards/total_composite/mean": 0.4260200307919429, "eval_rewards/total_composite/std": 0.3662373045316109, "eval_reward": 0.4260200307919429, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.02707064416832649, "eval_sampling/sampling_logp_difference/max": 0.9593030489408053, "eval_sampling/importance_sampling_ratio/min": 0.3921516973238725, "eval_sampling/importance_sampling_ratio/mean": 1.0070600234545195, "eval_sampling/importance_sampling_ratio/max": 1.4355276914743276, "eval_entropy": 0.26288481056690216, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.4260200307919429, "eval_reward_meter_mean": 0.5874437128122036, "eval_reward_meter_std": 0.4289410481086144, "eval_reward_count_adherence_mean": 0.8191676002282363, "eval_reward_count_adherence_std": 0.17260103730055001, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_repeat_penalty_mean": 0.8410639992127051, "eval_reward_repeat_penalty_std": 0.16562061241039863, "eval_reward_total_composite_mean": 0.4260200307919429, "eval_reward_total_composite_std": 0.3662373045316109} {"timestamp_utc": "2026-04-12T00:13:40Z", "mode": "train", "global_step": 1451, "epoch": 0.05828011406996827, "loss": -0.0183, "grad_norm": 9.900911331176758, "learning_rate": 5.606060606060606e-06, "num_tokens": 3266919.0, "completions/mean_length": 61.875, "completions/min_length": 55.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.875, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.5548748970031738, "rewards/meter/std": 0.3548371493816376, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5548748970031738, "rewards/total_composite/std": 0.3548371493816376, "reward": 0.5548748970031738, "reward_std": 0.3548371493816376, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12563803791999817, "sampling/sampling_logp_difference/max": 1.3014025688171387, "sampling/importance_sampling_ratio/min": 0.27214980125427246, "sampling/importance_sampling_ratio/mean": 1.0315823554992676, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.44280257076025, "clip_ratio/low_mean": 0.04481779085472226, "clip_ratio/low_min": 0.04481779085472226, "clip_ratio/high_mean": 0.04324009292759001, "clip_ratio/high_max": 0.04324009292759001, "clip_ratio/region_mean": 0.08805788378231227, "reward_total_mean": 0.5548748970031738, "reward_meter_mean": 0.5548748970031738, "reward_meter_std": 0.3548371493816376, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5548748970031738, "reward_total_composite_std": 0.3548371493816376} {"timestamp_utc": "2026-04-12T00:13:46Z", "mode": "train", "global_step": 1452, "epoch": 0.05832027955175322, "loss": 0.0264, "grad_norm": 3.927211046218872, "learning_rate": 5.603030303030303e-06, "num_tokens": 3269067.0, "completions/mean_length": 102.5, "completions/min_length": 95.0, "completions/max_length": 110.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.5, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 110.0, "rewards/meter/mean": 0.9900820255279541, "rewards/meter/std": 0.011413360014557838, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9900820255279541, "rewards/total_composite/std": 0.011413360014557838, "reward": 0.9900820255279541, "reward_std": 0.011413362808525562, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06290001422166824, "sampling/sampling_logp_difference/max": 1.65745210647583, "sampling/importance_sampling_ratio/min": 0.19062404334545135, "sampling/importance_sampling_ratio/mean": 1.0114924907684326, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5281493403017521, "clip_ratio/low_mean": 0.013940956210717559, "clip_ratio/low_min": 0.013940956210717559, "clip_ratio/high_mean": 0.041309412801638246, "clip_ratio/high_max": 0.041309412801638246, "clip_ratio/region_mean": 0.055250369012355804, "reward_total_mean": 0.9900820255279541, "reward_meter_mean": 0.9900820255279541, "reward_meter_std": 0.011413360014557838, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9900820255279541, "reward_total_composite_std": 0.011413360014557838} {"timestamp_utc": "2026-04-12T00:13:50Z", "mode": "train", "global_step": 1453, "epoch": 0.058360445033538176, "loss": 0.029, "grad_norm": 4.87343168258667, "learning_rate": 5.600000000000001e-06, "num_tokens": 3271008.0, "completions/mean_length": 61.625, "completions/min_length": 58.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.625, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.18945728242397308, "rewards/meter/std": 0.2888922691345215, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.18945728242397308, "rewards/total_composite/std": 0.2888922691345215, "reward": 0.18945728242397308, "reward_std": 0.28889229893684387, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06080813705921173, "sampling/sampling_logp_difference/max": 2.205195426940918, "sampling/importance_sampling_ratio/min": 0.11022898554801941, "sampling/importance_sampling_ratio/mean": 1.011013150215149, "sampling/importance_sampling_ratio/max": 1.7248963117599487, "entropy": 0.5127889458090067, "clip_ratio/low_mean": 0.024039480136707425, "clip_ratio/low_min": 0.024039480136707425, "clip_ratio/high_mean": 0.01592310261912644, "clip_ratio/high_max": 0.01592310261912644, "clip_ratio/region_mean": 0.039962582755833864, "reward_total_mean": 0.18945728242397308, "reward_meter_mean": 0.18945728242397308, "reward_meter_std": 0.2888922691345215, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.18945728242397308, "reward_total_composite_std": 0.2888922691345215} {"timestamp_utc": "2026-04-12T00:13:55Z", "mode": "train", "global_step": 1454, "epoch": 0.05840061051532313, "loss": 0.0102, "grad_norm": 3.9825429916381836, "learning_rate": 5.596969696969697e-06, "num_tokens": 3272821.0, "completions/mean_length": 58.625, "completions/min_length": 57.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.625, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9886520504951477, "rewards/meter/std": 0.006697438657283783, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9886520504951477, "rewards/total_composite/std": 0.006697438657283783, "reward": 0.9886520504951477, "reward_std": 0.00669743912294507, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03104584477841854, "sampling/sampling_logp_difference/max": 1.1052379608154297, "sampling/importance_sampling_ratio/min": 0.33113205432891846, "sampling/importance_sampling_ratio/mean": 0.9944877028465271, "sampling/importance_sampling_ratio/max": 1.4476277828216553, "entropy": 0.15899417083710432, "clip_ratio/low_mean": 0.004204352619126439, "clip_ratio/low_min": 0.004204352619126439, "clip_ratio/high_mean": 0.017170886043459177, "clip_ratio/high_max": 0.017170886043459177, "clip_ratio/region_mean": 0.021375238662585616, "reward_total_mean": 0.9886520504951477, "reward_meter_mean": 0.9886520504951477, "reward_meter_std": 0.006697438657283783, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9886520504951477, "reward_total_composite_std": 0.006697438657283783} {"timestamp_utc": "2026-04-12T00:13:59Z", "mode": "train", "global_step": 1455, "epoch": 0.058440775997108084, "loss": 0.0227, "grad_norm": 7.097043514251709, "learning_rate": 5.593939393939395e-06, "num_tokens": 3274349.0, "completions/mean_length": 36.0, "completions/min_length": 35.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9973074793815613, "rewards/meter/std": 0.0026173454243689775, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973074793815613, "rewards/total_composite/std": 0.0026173454243689775, "reward": 0.9973074793815613, "reward_std": 0.0026173454243689775, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04888688772916794, "sampling/sampling_logp_difference/max": 0.6760835647583008, "sampling/importance_sampling_ratio/min": 0.5086050033569336, "sampling/importance_sampling_ratio/mean": 0.9993876218795776, "sampling/importance_sampling_ratio/max": 1.5138667821884155, "entropy": 0.32455965876579285, "clip_ratio/low_mean": 0.0034722222480922937, "clip_ratio/low_min": 0.0034722222480922937, "clip_ratio/high_mean": 0.0423718374222517, "clip_ratio/high_max": 0.0423718374222517, "clip_ratio/region_mean": 0.045844059670343995, "reward_total_mean": 0.9973074793815613, "reward_meter_mean": 0.9973074793815613, "reward_meter_std": 0.0026173454243689775, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9973074793815613, "reward_total_composite_std": 0.0026173454243689775} {"timestamp_utc": "2026-04-12T00:14:07Z", "mode": "train", "global_step": 1456, "epoch": 0.05848094147889304, "loss": -0.0485, "grad_norm": 2.318981170654297, "learning_rate": 5.5909090909090915e-06, "num_tokens": 3278247.0, "completions/mean_length": 263.25, "completions/min_length": 234.0, "completions/max_length": 277.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 263.25, "completions/min_terminated_length": 234.0, "completions/max_terminated_length": 277.0, "rewards/meter/mean": 0.9962884187698364, "rewards/meter/std": 0.0015494668623432517, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.0578637570142746, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.911057710647583, "rewards/repeat_penalty/std": 0.051787521690130234, "rewards/total_composite/mean": 0.7669380903244019, "rewards/total_composite/std": 0.07985159009695053, "reward": 0.7669380903244019, "reward_std": 0.07985157519578934, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05095827206969261, "sampling/sampling_logp_difference/max": 1.4401264190673828, "sampling/importance_sampling_ratio/min": 0.2368977963924408, "sampling/importance_sampling_ratio/mean": 1.0123025178909302, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.41418151557445526, "clip_ratio/low_mean": 0.012334733735769987, "clip_ratio/low_min": 0.012334733735769987, "clip_ratio/high_mean": 0.03280176408588886, "clip_ratio/high_max": 0.03280176408588886, "clip_ratio/region_mean": 0.04513649782165885, "reward_total_mean": 0.7669380903244019, "reward_meter_mean": 0.9962884187698364, "reward_meter_std": 0.0015494668623432517, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.0578637570142746, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.911057710647583, "reward_repeat_penalty_std": 0.051787521690130234, "reward_total_composite_mean": 0.7669380903244019, "reward_total_composite_std": 0.07985159009695053} {"timestamp_utc": "2026-04-12T00:14:12Z", "mode": "train", "global_step": 1457, "epoch": 0.05852110696067799, "loss": -0.0297, "grad_norm": 2.4325199127197266, "learning_rate": 5.587878787878789e-06, "num_tokens": 3280349.0, "completions/mean_length": 96.75, "completions/min_length": 89.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.75, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.9892526268959045, "rewards/meter/std": 0.015538093633949757, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.2070196568965912, "rewards/total_composite/mean": 0.8396018743515015, "rewards/total_composite/std": 0.20004859566688538, "reward": 0.8396018743515015, "reward_std": 0.20004859566688538, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05370476841926575, "sampling/sampling_logp_difference/max": 2.0252418518066406, "sampling/importance_sampling_ratio/min": 0.13196192681789398, "sampling/importance_sampling_ratio/mean": 1.0023468732833862, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3380627855658531, "clip_ratio/low_mean": 0.00958982715383172, "clip_ratio/low_min": 0.00958982715383172, "clip_ratio/high_mean": 0.04439025931060314, "clip_ratio/high_max": 0.04439025931060314, "clip_ratio/region_mean": 0.05398008646443486, "reward_total_mean": 0.8396018743515015, "reward_meter_mean": 0.9892526268959045, "reward_meter_std": 0.015538093633949757, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.2070196568965912, "reward_total_composite_mean": 0.8396018743515015, "reward_total_composite_std": 0.20004859566688538} {"timestamp_utc": "2026-04-12T00:14:17Z", "mode": "train", "global_step": 1458, "epoch": 0.058561272442462946, "loss": -0.0117, "grad_norm": 4.565279006958008, "learning_rate": 5.584848484848485e-06, "num_tokens": 3282489.0, "completions/mean_length": 93.5, "completions/min_length": 89.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.5, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9970370531082153, "rewards/meter/std": 0.0014831003500148654, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7750000357627869, "rewards/repeat_penalty/std": 0.16690459847450256, "rewards/total_composite/mean": 0.772584080696106, "rewards/total_composite/std": 0.1656908094882965, "reward": 0.772584080696106, "reward_std": 0.1656908094882965, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03978914022445679, "sampling/sampling_logp_difference/max": 1.165329098701477, "sampling/importance_sampling_ratio/min": 0.31182003021240234, "sampling/importance_sampling_ratio/mean": 1.0047050714492798, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.24885553307831287, "clip_ratio/low_mean": 0.009437821572646499, "clip_ratio/low_min": 0.009437821572646499, "clip_ratio/high_mean": 0.022278774995356798, "clip_ratio/high_max": 0.022278774995356798, "clip_ratio/region_mean": 0.0317165965680033, "reward_total_mean": 0.772584080696106, "reward_meter_mean": 0.9970370531082153, "reward_meter_std": 0.0014831003500148654, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7750000357627869, "reward_repeat_penalty_std": 0.16690459847450256, "reward_total_composite_mean": 0.772584080696106, "reward_total_composite_std": 0.1656908094882965} {"timestamp_utc": "2026-04-12T00:14:25Z", "mode": "train", "global_step": 1459, "epoch": 0.0586014379242479, "loss": -0.0396, "grad_norm": 3.691830635070801, "learning_rate": 5.5818181818181824e-06, "num_tokens": 3285975.0, "completions/mean_length": 223.75, "completions/min_length": 200.0, "completions/max_length": 258.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 223.75, "completions/min_terminated_length": 200.0, "completions/max_terminated_length": 258.0, "rewards/meter/mean": 0.2490365207195282, "rewards/meter/std": 0.21091364324092865, "rewards/count_adherence/mean": 0.5576923489570618, "rewards/count_adherence/std": 0.05439283326268196, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8426282405853271, "rewards/repeat_penalty/std": 0.1708158552646637, "rewards/total_composite/mean": 0.12107732146978378, "rewards/total_composite/std": 0.09244445711374283, "reward": 0.12107732146978378, "reward_std": 0.09244445711374283, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0653611496090889, "sampling/sampling_logp_difference/max": 1.6574034690856934, "sampling/importance_sampling_ratio/min": 0.19063332676887512, "sampling/importance_sampling_ratio/mean": 1.0115958452224731, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5131779182702303, "clip_ratio/low_mean": 0.014779910678043962, "clip_ratio/low_min": 0.014779910678043962, "clip_ratio/high_mean": 0.03423361713066697, "clip_ratio/high_max": 0.03423361713066697, "clip_ratio/region_mean": 0.04901352780871093, "reward_total_mean": 0.12107732146978378, "reward_meter_mean": 0.2490365207195282, "reward_meter_std": 0.21091364324092865, "reward_count_adherence_mean": 0.5576923489570618, "reward_count_adherence_std": 0.05439283326268196, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8426282405853271, "reward_repeat_penalty_std": 0.1708158552646637, "reward_total_composite_mean": 0.12107732146978378, "reward_total_composite_std": 0.09244445711374283} {"timestamp_utc": "2026-04-12T00:14:29Z", "mode": "train", "global_step": 1460, "epoch": 0.058641603406032854, "loss": 0.0048, "grad_norm": 4.334770679473877, "learning_rate": 5.578787878787879e-06, "num_tokens": 3287646.0, "completions/mean_length": 65.875, "completions/min_length": 65.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.875, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9952852129936218, "rewards/meter/std": 0.0014361762441694736, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9952852129936218, "rewards/total_composite/std": 0.0014361762441694736, "reward": 0.9952852129936218, "reward_std": 0.0014361703069880605, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03258511424064636, "sampling/sampling_logp_difference/max": 1.3955812454223633, "sampling/importance_sampling_ratio/min": 0.24768902361392975, "sampling/importance_sampling_ratio/mean": 1.0027287006378174, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1356327636167407, "clip_ratio/low_mean": 0.003759611048735678, "clip_ratio/low_min": 0.003759611048735678, "clip_ratio/high_mean": 0.026631702319718897, "clip_ratio/high_max": 0.026631702319718897, "clip_ratio/region_mean": 0.030391313368454576, "reward_total_mean": 0.9952852129936218, "reward_meter_mean": 0.9952852129936218, "reward_meter_std": 0.0014361762441694736, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9952852129936218, "reward_total_composite_std": 0.0014361762441694736} {"timestamp_utc": "2026-04-12T00:14:34Z", "mode": "train", "global_step": 1461, "epoch": 0.05868176888781781, "loss": 0.0375, "grad_norm": 7.510387420654297, "learning_rate": 5.575757575757577e-06, "num_tokens": 3289631.0, "completions/mean_length": 68.125, "completions/min_length": 66.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.125, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.8637334108352661, "rewards/meter/std": 0.34701576828956604, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8637334108352661, "rewards/total_composite/std": 0.34701576828956604, "reward": 0.8637334108352661, "reward_std": 0.34701573848724365, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03330973535776138, "sampling/sampling_logp_difference/max": 1.0436010360717773, "sampling/importance_sampling_ratio/min": 0.3521841764450073, "sampling/importance_sampling_ratio/mean": 1.0041791200637817, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.24057043716311455, "clip_ratio/low_mean": 0.008333333767950535, "clip_ratio/low_min": 0.008333333767950535, "clip_ratio/high_mean": 0.016740256920456886, "clip_ratio/high_max": 0.016740256920456886, "clip_ratio/region_mean": 0.02507359068840742, "reward_total_mean": 0.8637334108352661, "reward_meter_mean": 0.8637334108352661, "reward_meter_std": 0.34701576828956604, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8637334108352661, "reward_total_composite_std": 0.34701576828956604} {"timestamp_utc": "2026-04-12T00:14:39Z", "mode": "train", "global_step": 1462, "epoch": 0.05872193436960276, "loss": 0.0109, "grad_norm": 5.110687255859375, "learning_rate": 5.572727272727273e-06, "num_tokens": 3291531.0, "completions/mean_length": 69.5, "completions/min_length": 65.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.5, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9955259561538696, "rewards/meter/std": 0.0017520119436085224, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9955259561538696, "rewards/total_composite/std": 0.0017520119436085224, "reward": 0.9955259561538696, "reward_std": 0.001752000767737627, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06455224007368088, "sampling/sampling_logp_difference/max": 1.3436126708984375, "sampling/importance_sampling_ratio/min": 0.26090142130851746, "sampling/importance_sampling_ratio/mean": 1.018951654434204, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5490291193127632, "clip_ratio/low_mean": 0.029018934816122055, "clip_ratio/low_min": 0.029018934816122055, "clip_ratio/high_mean": 0.02201478136703372, "clip_ratio/high_max": 0.02201478136703372, "clip_ratio/region_mean": 0.051033716183155775, "reward_total_mean": 0.9955259561538696, "reward_meter_mean": 0.9955259561538696, "reward_meter_std": 0.0017520119436085224, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9955259561538696, "reward_total_composite_std": 0.0017520119436085224} {"timestamp_utc": "2026-04-12T00:14:43Z", "mode": "train", "global_step": 1463, "epoch": 0.058762099851387715, "loss": -0.0018, "grad_norm": 5.067295551300049, "learning_rate": 5.569696969696971e-06, "num_tokens": 3292987.0, "completions/mean_length": 63.0, "completions/min_length": 59.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.0, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.7616782188415527, "rewards/meter/std": 0.24172598123550415, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7616782188415527, "rewards/total_composite/std": 0.24172598123550415, "reward": 0.7616782188415527, "reward_std": 0.24172596633434296, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07128675282001495, "sampling/sampling_logp_difference/max": 1.4407329559326172, "sampling/importance_sampling_ratio/min": 0.2367541640996933, "sampling/importance_sampling_ratio/mean": 1.0115406513214111, "sampling/importance_sampling_ratio/max": 1.8223750591278076, "entropy": 0.5481097176671028, "clip_ratio/low_mean": 0.027423894964158535, "clip_ratio/low_min": 0.027423894964158535, "clip_ratio/high_mean": 0.02786910650320351, "clip_ratio/high_max": 0.02786910650320351, "clip_ratio/region_mean": 0.055293001467362046, "reward_total_mean": 0.7616782188415527, "reward_meter_mean": 0.7616782188415527, "reward_meter_std": 0.24172598123550415, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7616782188415527, "reward_total_composite_std": 0.24172598123550415} {"timestamp_utc": "2026-04-12T00:14:48Z", "mode": "train", "global_step": 1464, "epoch": 0.05880226533317267, "loss": -0.0081, "grad_norm": 5.747054100036621, "learning_rate": 5.566666666666667e-06, "num_tokens": 3294759.0, "completions/mean_length": 59.5, "completions/min_length": 57.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.5, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.33728209137916565, "rewards/meter/std": 0.2937431037425995, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.33728209137916565, "rewards/total_composite/std": 0.2937431037425995, "reward": 0.33728209137916565, "reward_std": 0.2937430739402771, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06646484136581421, "sampling/sampling_logp_difference/max": 0.8989553451538086, "sampling/importance_sampling_ratio/min": 0.40699464082717896, "sampling/importance_sampling_ratio/mean": 1.0084482431411743, "sampling/importance_sampling_ratio/max": 1.9918129444122314, "entropy": 0.5435944311320782, "clip_ratio/low_mean": 0.01695890142582357, "clip_ratio/low_min": 0.01695890142582357, "clip_ratio/high_mean": 0.02262367820367217, "clip_ratio/high_max": 0.02262367820367217, "clip_ratio/region_mean": 0.03958257962949574, "reward_total_mean": 0.33728209137916565, "reward_meter_mean": 0.33728209137916565, "reward_meter_std": 0.2937431037425995, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.33728209137916565, "reward_total_composite_std": 0.2937431037425995} {"timestamp_utc": "2026-04-12T00:14:53Z", "mode": "train", "global_step": 1465, "epoch": 0.05884243081495762, "loss": 0.0259, "grad_norm": 5.774188041687012, "learning_rate": 5.563636363636364e-06, "num_tokens": 3296618.0, "completions/mean_length": 67.375, "completions/min_length": 65.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.375, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9969590306282043, "rewards/meter/std": 0.0012544745113700628, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9969590306282043, "rewards/total_composite/std": 0.0012544745113700628, "reward": 0.9969590306282043, "reward_std": 0.0012544887140393257, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06404515355825424, "sampling/sampling_logp_difference/max": 2.009974479675293, "sampling/importance_sampling_ratio/min": 0.13399209082126617, "sampling/importance_sampling_ratio/mean": 1.0145848989486694, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4691624492406845, "clip_ratio/low_mean": 0.010546227567829192, "clip_ratio/low_min": 0.010546227567829192, "clip_ratio/high_mean": 0.02795016940217465, "clip_ratio/high_max": 0.02795016940217465, "clip_ratio/region_mean": 0.03849639697000384, "reward_total_mean": 0.9969590306282043, "reward_meter_mean": 0.9969590306282043, "reward_meter_std": 0.0012544745113700628, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9969590306282043, "reward_total_composite_std": 0.0012544745113700628} {"timestamp_utc": "2026-04-12T00:14:57Z", "mode": "train", "global_step": 1466, "epoch": 0.05888259629674258, "loss": 0.0091, "grad_norm": 6.224666118621826, "learning_rate": 5.560606060606061e-06, "num_tokens": 3298395.0, "completions/mean_length": 59.125, "completions/min_length": 56.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.125, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9083940982818604, "rewards/meter/std": 0.1951943337917328, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9083940982818604, "rewards/total_composite/std": 0.1951943337917328, "reward": 0.9083940982818604, "reward_std": 0.1951943188905716, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030300971120595932, "sampling/sampling_logp_difference/max": 0.9119729995727539, "sampling/importance_sampling_ratio/min": 0.40173083543777466, "sampling/importance_sampling_ratio/mean": 1.0091822147369385, "sampling/importance_sampling_ratio/max": 1.7400184869766235, "entropy": 0.2567995712161064, "clip_ratio/low_mean": 0.0020833334419876337, "clip_ratio/low_min": 0.0020833334419876337, "clip_ratio/high_mean": 0.025272560073062778, "clip_ratio/high_max": 0.025272560073062778, "clip_ratio/region_mean": 0.02735589351505041, "reward_total_mean": 0.9083940982818604, "reward_meter_mean": 0.9083940982818604, "reward_meter_std": 0.1951943337917328, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9083940982818604, "reward_total_composite_std": 0.1951943337917328} {"timestamp_utc": "2026-04-12T00:15:04Z", "mode": "train", "global_step": 1467, "epoch": 0.05892276177852753, "loss": -0.0184, "grad_norm": 3.2179627418518066, "learning_rate": 5.557575757575758e-06, "num_tokens": 3301728.0, "completions/mean_length": 220.625, "completions/min_length": 201.0, "completions/max_length": 236.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 220.625, "completions/min_terminated_length": 201.0, "completions/max_terminated_length": 236.0, "rewards/meter/mean": 0.3030664324760437, "rewards/meter/std": 0.29540300369262695, "rewards/count_adherence/mean": 0.5288461446762085, "rewards/count_adherence/std": 0.027196424081921577, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8133013248443604, "rewards/repeat_penalty/std": 0.10961552709341049, "rewards/total_composite/mean": 0.1352500319480896, "rewards/total_composite/std": 0.13207849860191345, "reward": 0.1352500319480896, "reward_std": 0.13207849860191345, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05384691804647446, "sampling/sampling_logp_difference/max": 1.9810237884521484, "sampling/importance_sampling_ratio/min": 0.13792794942855835, "sampling/importance_sampling_ratio/mean": 1.006094217300415, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3614166211336851, "clip_ratio/low_mean": 0.024785209679976106, "clip_ratio/low_min": 0.024785209679976106, "clip_ratio/high_mean": 0.014436864759773016, "clip_ratio/high_max": 0.014436864759773016, "clip_ratio/region_mean": 0.03922207443974912, "reward_total_mean": 0.1352500319480896, "reward_meter_mean": 0.3030664324760437, "reward_meter_std": 0.29540300369262695, "reward_count_adherence_mean": 0.5288461446762085, "reward_count_adherence_std": 0.027196424081921577, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8133013248443604, "reward_repeat_penalty_std": 0.10961552709341049, "reward_total_composite_mean": 0.1352500319480896, "reward_total_composite_std": 0.13207849860191345} {"timestamp_utc": "2026-04-12T00:15:09Z", "mode": "train", "global_step": 1468, "epoch": 0.058962927260312485, "loss": 0.0394, "grad_norm": 3.850555419921875, "learning_rate": 5.554545454545454e-06, "num_tokens": 3303612.0, "completions/mean_length": 70.5, "completions/min_length": 62.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.5, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.993515133857727, "rewards/meter/std": 0.002871233271434903, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.993515133857727, "rewards/total_composite/std": 0.002871233271434903, "reward": 0.993515133857727, "reward_std": 0.0028712216299027205, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06810689717531204, "sampling/sampling_logp_difference/max": 1.0864152908325195, "sampling/importance_sampling_ratio/min": 0.3374238908290863, "sampling/importance_sampling_ratio/mean": 1.0134084224700928, "sampling/importance_sampling_ratio/max": 1.9790068864822388, "entropy": 0.5614041350781918, "clip_ratio/low_mean": 0.0190093262353912, "clip_ratio/low_min": 0.0190093262353912, "clip_ratio/high_mean": 0.023276393418200314, "clip_ratio/high_max": 0.023276393418200314, "clip_ratio/region_mean": 0.042285719653591514, "reward_total_mean": 0.993515133857727, "reward_meter_mean": 0.993515133857727, "reward_meter_std": 0.002871233271434903, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.993515133857727, "reward_total_composite_std": 0.002871233271434903} {"timestamp_utc": "2026-04-12T00:15:14Z", "mode": "train", "global_step": 1469, "epoch": 0.05900309274209744, "loss": -0.0341, "grad_norm": 3.2438156604766846, "learning_rate": 5.5515151515151524e-06, "num_tokens": 3305697.0, "completions/mean_length": 92.625, "completions/min_length": 87.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.625, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.7984445691108704, "rewards/meter/std": 0.2091599404811859, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.17728103697299957, "rewards/total_composite/mean": 0.7041343450546265, "rewards/total_composite/std": 0.271832674741745, "reward": 0.7041343450546265, "reward_std": 0.2718327045440674, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.041127223521471024, "sampling/sampling_logp_difference/max": 1.2640190124511719, "sampling/importance_sampling_ratio/min": 0.2825163006782532, "sampling/importance_sampling_ratio/mean": 1.0063135623931885, "sampling/importance_sampling_ratio/max": 1.4045096635818481, "entropy": 0.2813885472714901, "clip_ratio/low_mean": 0.0071839080192148685, "clip_ratio/low_min": 0.0071839080192148685, "clip_ratio/high_mean": 0.025160838733427227, "clip_ratio/high_max": 0.025160838733427227, "clip_ratio/region_mean": 0.032344746752642095, "reward_total_mean": 0.7041343450546265, "reward_meter_mean": 0.7984445691108704, "reward_meter_std": 0.2091599404811859, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.17728103697299957, "reward_total_composite_mean": 0.7041343450546265, "reward_total_composite_std": 0.271832674741745} {"timestamp_utc": "2026-04-12T00:15:19Z", "mode": "train", "global_step": 1470, "epoch": 0.05904325822388239, "loss": -0.0266, "grad_norm": 6.198017120361328, "learning_rate": 5.548484848484849e-06, "num_tokens": 3307476.0, "completions/mean_length": 70.375, "completions/min_length": 64.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.375, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.8553205132484436, "rewards/meter/std": 0.22217348217964172, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8553205132484436, "rewards/total_composite/std": 0.22217348217964172, "reward": 0.8553205132484436, "reward_std": 0.22217348217964172, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07493096590042114, "sampling/sampling_logp_difference/max": 1.1873397827148438, "sampling/importance_sampling_ratio/min": 0.3050316274166107, "sampling/importance_sampling_ratio/mean": 1.0185297727584839, "sampling/importance_sampling_ratio/max": 1.9982537031173706, "entropy": 0.7106457650661469, "clip_ratio/low_mean": 0.013322061393409967, "clip_ratio/low_min": 0.013322061393409967, "clip_ratio/high_mean": 0.04982595471665263, "clip_ratio/high_max": 0.04982595471665263, "clip_ratio/region_mean": 0.0631480161100626, "reward_total_mean": 0.8553205132484436, "reward_meter_mean": 0.8553205132484436, "reward_meter_std": 0.22217348217964172, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8553205132484436, "reward_total_composite_std": 0.22217348217964172} {"timestamp_utc": "2026-04-12T00:15:23Z", "mode": "train", "global_step": 1471, "epoch": 0.05908342370566735, "loss": -0.0016, "grad_norm": 4.865560054779053, "learning_rate": 5.545454545454546e-06, "num_tokens": 3309226.0, "completions/mean_length": 56.75, "completions/min_length": 53.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.75, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9437645673751831, "rewards/meter/std": 0.12552663683891296, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9023252725601196, "rewards/total_composite/std": 0.15711426734924316, "reward": 0.9023252725601196, "reward_std": 0.15711423754692078, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03652970865368843, "sampling/sampling_logp_difference/max": 1.6262216567993164, "sampling/importance_sampling_ratio/min": 0.196671262383461, "sampling/importance_sampling_ratio/mean": 1.0094138383865356, "sampling/importance_sampling_ratio/max": 1.7097090482711792, "entropy": 0.237571744248271, "clip_ratio/low_mean": 0.011254789307713509, "clip_ratio/low_min": 0.011254789307713509, "clip_ratio/high_mean": 0.021573547972366214, "clip_ratio/high_max": 0.021573547972366214, "clip_ratio/region_mean": 0.03282833728007972, "reward_total_mean": 0.9023252725601196, "reward_meter_mean": 0.9437645673751831, "reward_meter_std": 0.12552663683891296, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.9023252725601196, "reward_total_composite_std": 0.15711426734924316} {"timestamp_utc": "2026-04-12T00:15:29Z", "mode": "train", "global_step": 1472, "epoch": 0.0591235891874523, "loss": -0.0156, "grad_norm": 2.8264143466949463, "learning_rate": 5.5424242424242425e-06, "num_tokens": 3311622.0, "completions/mean_length": 123.5, "completions/min_length": 114.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 123.5, "completions/min_terminated_length": 114.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.7211212515830994, "rewards/meter/std": 0.2165415734052658, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7321428060531616, "rewards/repeat_penalty/std": 0.1937432438135147, "rewards/total_composite/mean": 0.5241959095001221, "rewards/total_composite/std": 0.20148435235023499, "reward": 0.5241959095001221, "reward_std": 0.20148435235023499, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03288319706916809, "sampling/sampling_logp_difference/max": 1.2717273235321045, "sampling/importance_sampling_ratio/min": 0.28034695982933044, "sampling/importance_sampling_ratio/mean": 1.0071460008621216, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.21886500529944897, "clip_ratio/low_mean": 0.005906250094994903, "clip_ratio/low_min": 0.005906250094994903, "clip_ratio/high_mean": 0.009076492046006024, "clip_ratio/high_max": 0.009076492046006024, "clip_ratio/region_mean": 0.014982742141000926, "reward_total_mean": 0.5241959095001221, "reward_meter_mean": 0.7211212515830994, "reward_meter_std": 0.2165415734052658, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7321428060531616, "reward_repeat_penalty_std": 0.1937432438135147, "reward_total_composite_mean": 0.5241959095001221, "reward_total_composite_std": 0.20148435235023499} {"timestamp_utc": "2026-04-12T00:15:35Z", "mode": "train", "global_step": 1473, "epoch": 0.059163754669237255, "loss": 0.0149, "grad_norm": 2.3183512687683105, "learning_rate": 5.53939393939394e-06, "num_tokens": 3314535.0, "completions/mean_length": 184.125, "completions/min_length": 164.0, "completions/max_length": 202.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 184.125, "completions/min_terminated_length": 164.0, "completions/max_terminated_length": 202.0, "rewards/meter/mean": 0.8684302568435669, "rewards/meter/std": 0.16792094707489014, "rewards/count_adherence/mean": 0.7857142686843872, "rewards/count_adherence/std": 0.07636035233736038, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7905303239822388, "rewards/repeat_penalty/std": 0.09503685683012009, "rewards/total_composite/mean": 0.5370750427246094, "rewards/total_composite/std": 0.12571245431900024, "reward": 0.5370750427246094, "reward_std": 0.12571245431900024, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022778350859880447, "sampling/sampling_logp_difference/max": 1.592844009399414, "sampling/importance_sampling_ratio/min": 0.20334647595882416, "sampling/importance_sampling_ratio/mean": 1.0009946823120117, "sampling/importance_sampling_ratio/max": 1.7813695669174194, "entropy": 0.12613181956112385, "clip_ratio/low_mean": 0.005232378258369863, "clip_ratio/low_min": 0.005232378258369863, "clip_ratio/high_mean": 0.006139094999525696, "clip_ratio/high_max": 0.006139094999525696, "clip_ratio/region_mean": 0.011371473257895559, "reward_total_mean": 0.5370750427246094, "reward_meter_mean": 0.8684302568435669, "reward_meter_std": 0.16792094707489014, "reward_count_adherence_mean": 0.7857142686843872, "reward_count_adherence_std": 0.07636035233736038, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7905303239822388, "reward_repeat_penalty_std": 0.09503685683012009, "reward_total_composite_mean": 0.5370750427246094, "reward_total_composite_std": 0.12571245431900024} {"timestamp_utc": "2026-04-12T00:15:41Z", "mode": "train", "global_step": 1474, "epoch": 0.05920392015102221, "loss": 0.0044, "grad_norm": 2.7291924953460693, "learning_rate": 5.536363636363636e-06, "num_tokens": 3317453.0, "completions/mean_length": 166.75, "completions/min_length": 162.0, "completions/max_length": 170.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 166.75, "completions/min_terminated_length": 162.0, "completions/max_terminated_length": 170.0, "rewards/meter/mean": 0.959197998046875, "rewards/meter/std": 0.10039316117763519, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8055555820465088, "rewards/repeat_penalty/std": 0.11501092463731766, "rewards/total_composite/mean": 0.6380866169929504, "rewards/total_composite/std": 0.07123729586601257, "reward": 0.6380866169929504, "reward_std": 0.07123728096485138, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027119241654872894, "sampling/sampling_logp_difference/max": 1.337611198425293, "sampling/importance_sampling_ratio/min": 0.26247191429138184, "sampling/importance_sampling_ratio/mean": 1.0012325048446655, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15955681633204222, "clip_ratio/low_mean": 0.008916479535400867, "clip_ratio/low_min": 0.008916479535400867, "clip_ratio/high_mean": 0.018076194741297513, "clip_ratio/high_max": 0.018076194741297513, "clip_ratio/region_mean": 0.02699267427669838, "reward_total_mean": 0.6380866169929504, "reward_meter_mean": 0.959197998046875, "reward_meter_std": 0.10039316117763519, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8055555820465088, "reward_repeat_penalty_std": 0.11501092463731766, "reward_total_composite_mean": 0.6380866169929504, "reward_total_composite_std": 0.07123729586601257} {"timestamp_utc": "2026-04-12T00:15:46Z", "mode": "train", "global_step": 1475, "epoch": 0.05924408563280716, "loss": 0.0032, "grad_norm": 7.203929424285889, "learning_rate": 5.533333333333334e-06, "num_tokens": 3319564.0, "completions/mean_length": 102.875, "completions/min_length": 96.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.875, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.9892958402633667, "rewards/meter/std": 0.02243867516517639, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9892958402633667, "rewards/total_composite/std": 0.02243867516517639, "reward": 0.9892958402633667, "reward_std": 0.022438665851950645, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06217264011502266, "sampling/sampling_logp_difference/max": 2.3590099811553955, "sampling/importance_sampling_ratio/min": 0.09451375156641006, "sampling/importance_sampling_ratio/mean": 1.0040569305419922, "sampling/importance_sampling_ratio/max": 1.710269570350647, "entropy": 0.4008924439549446, "clip_ratio/low_mean": 0.0035714285913854837, "clip_ratio/low_min": 0.0035714285913854837, "clip_ratio/high_mean": 0.044174039736390114, "clip_ratio/high_max": 0.044174039736390114, "clip_ratio/region_mean": 0.0477454683277756, "reward_total_mean": 0.9892958402633667, "reward_meter_mean": 0.9892958402633667, "reward_meter_std": 0.02243867516517639, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9892958402633667, "reward_total_composite_std": 0.02243867516517639} {"timestamp_utc": "2026-04-12T00:15:51Z", "mode": "train", "global_step": 1476, "epoch": 0.05928425111459212, "loss": -0.0149, "grad_norm": 4.188331604003906, "learning_rate": 5.530303030303031e-06, "num_tokens": 3321999.0, "completions/mean_length": 120.375, "completions/min_length": 113.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.375, "completions/min_terminated_length": 113.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.7211596369743347, "rewards/meter/std": 0.29947173595428467, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8214285373687744, "rewards/repeat_penalty/std": 0.1266293078660965, "rewards/total_composite/mean": 0.5741269588470459, "rewards/total_composite/std": 0.22667214274406433, "reward": 0.5741269588470459, "reward_std": 0.22667212784290314, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05203632265329361, "sampling/sampling_logp_difference/max": 1.7595195770263672, "sampling/importance_sampling_ratio/min": 0.2374485731124878, "sampling/importance_sampling_ratio/mean": 1.0193830728530884, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45302187465131283, "clip_ratio/low_mean": 0.013878762954846025, "clip_ratio/low_min": 0.013878762954846025, "clip_ratio/high_mean": 0.015247002593241632, "clip_ratio/high_max": 0.015247002593241632, "clip_ratio/region_mean": 0.029125765548087656, "reward_total_mean": 0.5741269588470459, "reward_meter_mean": 0.7211596369743347, "reward_meter_std": 0.29947173595428467, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8214285373687744, "reward_repeat_penalty_std": 0.1266293078660965, "reward_total_composite_mean": 0.5741269588470459, "reward_total_composite_std": 0.22667214274406433} {"timestamp_utc": "2026-04-12T00:15:56Z", "mode": "train", "global_step": 1477, "epoch": 0.05932441659637707, "loss": 0.0153, "grad_norm": 6.804332733154297, "learning_rate": 5.527272727272728e-06, "num_tokens": 3323777.0, "completions/mean_length": 68.25, "completions/min_length": 65.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.25, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9923118352890015, "rewards/meter/std": 0.008320425637066364, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9923118352890015, "rewards/total_composite/std": 0.008320425637066364, "reward": 0.9923118352890015, "reward_std": 0.008320420049130917, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0801224485039711, "sampling/sampling_logp_difference/max": 1.6460561752319336, "sampling/importance_sampling_ratio/min": 0.19280880689620972, "sampling/importance_sampling_ratio/mean": 1.0044664144515991, "sampling/importance_sampling_ratio/max": 1.8288415670394897, "entropy": 0.6124710254371166, "clip_ratio/low_mean": 0.01433747448027134, "clip_ratio/low_min": 0.01433747448027134, "clip_ratio/high_mean": 0.055169664323329926, "clip_ratio/high_max": 0.055169664323329926, "clip_ratio/region_mean": 0.06950713880360126, "reward_total_mean": 0.9923118352890015, "reward_meter_mean": 0.9923118352890015, "reward_meter_std": 0.008320425637066364, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9923118352890015, "reward_total_composite_std": 0.008320425637066364} {"timestamp_utc": "2026-04-12T00:16:00Z", "mode": "train", "global_step": 1478, "epoch": 0.059364582078162025, "loss": 0.0228, "grad_norm": 10.884235382080078, "learning_rate": 5.524242424242424e-06, "num_tokens": 3325231.0, "completions/mean_length": 29.75, "completions/min_length": 29.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.75, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9929543733596802, "rewards/meter/std": 0.0024007554166018963, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9929543733596802, "rewards/total_composite/std": 0.0024007554166018963, "reward": 0.9929543733596802, "reward_std": 0.002400748198851943, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04087400063872337, "sampling/sampling_logp_difference/max": 1.8403053283691406, "sampling/importance_sampling_ratio/min": 0.1587689369916916, "sampling/importance_sampling_ratio/mean": 1.0106008052825928, "sampling/importance_sampling_ratio/max": 1.9975814819335938, "entropy": 0.25689505971968174, "clip_ratio/low_mean": 0.012122844811528921, "clip_ratio/low_min": 0.012122844811528921, "clip_ratio/high_mean": 0.012526939623057842, "clip_ratio/high_max": 0.012526939623057842, "clip_ratio/region_mean": 0.024649784434586763, "reward_total_mean": 0.9929543733596802, "reward_meter_mean": 0.9929543733596802, "reward_meter_std": 0.0024007554166018963, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9929543733596802, "reward_total_composite_std": 0.0024007554166018963} {"timestamp_utc": "2026-04-12T00:16:05Z", "mode": "train", "global_step": 1479, "epoch": 0.05940474755994698, "loss": 0.021, "grad_norm": 6.581116676330566, "learning_rate": 5.521212121212122e-06, "num_tokens": 3327045.0, "completions/mean_length": 67.75, "completions/min_length": 63.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.75, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.8861526846885681, "rewards/meter/std": 0.3106532394886017, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8861526846885681, "rewards/total_composite/std": 0.3106532394886017, "reward": 0.8861526846885681, "reward_std": 0.3106532096862793, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06694039702415466, "sampling/sampling_logp_difference/max": 0.8918218612670898, "sampling/importance_sampling_ratio/min": 0.4099082946777344, "sampling/importance_sampling_ratio/mean": 1.0218294858932495, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6052872687578201, "clip_ratio/low_mean": 0.0036231884732842445, "clip_ratio/low_min": 0.0036231884732842445, "clip_ratio/high_mean": 0.052147963899187744, "clip_ratio/high_max": 0.052147963899187744, "clip_ratio/region_mean": 0.05577115237247199, "reward_total_mean": 0.8861526846885681, "reward_meter_mean": 0.8861526846885681, "reward_meter_std": 0.3106532394886017, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8861526846885681, "reward_total_composite_std": 0.3106532394886017} {"timestamp_utc": "2026-04-12T00:16:09Z", "mode": "train", "global_step": 1480, "epoch": 0.05944491304173193, "loss": 0.0211, "grad_norm": 12.08879280090332, "learning_rate": 5.518181818181818e-06, "num_tokens": 3328501.0, "completions/mean_length": 37.0, "completions/min_length": 35.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.997024655342102, "rewards/meter/std": 0.001989408629015088, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.997024655342102, "rewards/total_composite/std": 0.001989408629015088, "reward": 0.997024655342102, "reward_std": 0.001989412121474743, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06934916973114014, "sampling/sampling_logp_difference/max": 1.2974138259887695, "sampling/importance_sampling_ratio/min": 0.27323752641677856, "sampling/importance_sampling_ratio/mean": 1.000410795211792, "sampling/importance_sampling_ratio/max": 1.6434557437896729, "entropy": 0.5856506302952766, "clip_ratio/low_mean": 0.027099420549347997, "clip_ratio/low_min": 0.027099420549347997, "clip_ratio/high_mean": 0.023252843879163265, "clip_ratio/high_max": 0.023252843879163265, "clip_ratio/region_mean": 0.05035226442851126, "reward_total_mean": 0.997024655342102, "reward_meter_mean": 0.997024655342102, "reward_meter_std": 0.001989408629015088, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.997024655342102, "reward_total_composite_std": 0.001989408629015088} {"timestamp_utc": "2026-04-12T00:16:17Z", "mode": "train", "global_step": 1481, "epoch": 0.059485078523516886, "loss": -0.0133, "grad_norm": 3.2796733379364014, "learning_rate": 5.515151515151515e-06, "num_tokens": 3332149.0, "completions/mean_length": 243.0, "completions/min_length": 225.0, "completions/max_length": 267.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 243.0, "completions/min_terminated_length": 225.0, "completions/max_terminated_length": 267.0, "rewards/meter/mean": 0.7210352420806885, "rewards/meter/std": 0.4204363524913788, "rewards/count_adherence/mean": 0.5089285373687744, "rewards/count_adherence/std": 0.025253823027014732, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.848809540271759, "rewards/repeat_penalty/std": 0.11776864528656006, "rewards/total_composite/mean": 0.2932712435722351, "rewards/total_composite/std": 0.1669246405363083, "reward": 0.2932712435722351, "reward_std": 0.1669246405363083, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06317514181137085, "sampling/sampling_logp_difference/max": 1.6376686096191406, "sampling/importance_sampling_ratio/min": 0.1944328099489212, "sampling/importance_sampling_ratio/mean": 1.0123441219329834, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5313003696501255, "clip_ratio/low_mean": 0.015805913135409355, "clip_ratio/low_min": 0.015805913135409355, "clip_ratio/high_mean": 0.023832474602386355, "clip_ratio/high_max": 0.023832474602386355, "clip_ratio/region_mean": 0.03963838773779571, "reward_total_mean": 0.2932712435722351, "reward_meter_mean": 0.7210352420806885, "reward_meter_std": 0.4204363524913788, "reward_count_adherence_mean": 0.5089285373687744, "reward_count_adherence_std": 0.025253823027014732, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.848809540271759, "reward_repeat_penalty_std": 0.11776864528656006, "reward_total_composite_mean": 0.2932712435722351, "reward_total_composite_std": 0.1669246405363083} {"timestamp_utc": "2026-04-12T00:16:24Z", "mode": "train", "global_step": 1482, "epoch": 0.05952524400530184, "loss": -0.0162, "grad_norm": 4.245050430297852, "learning_rate": 5.512121212121213e-06, "num_tokens": 3334931.0, "completions/mean_length": 157.75, "completions/min_length": 135.0, "completions/max_length": 172.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 157.75, "completions/min_terminated_length": 135.0, "completions/max_terminated_length": 172.0, "rewards/meter/mean": 0.9921485781669617, "rewards/meter/std": 0.002238625893369317, "rewards/count_adherence/mean": 0.8035714030265808, "rewards/count_adherence/std": 0.07393559068441391, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8219696879386902, "rewards/repeat_penalty/std": 0.08952570706605911, "rewards/total_composite/mean": 0.6518079042434692, "rewards/total_composite/std": 0.05643462389707565, "reward": 0.6518079042434692, "reward_std": 0.05643462389707565, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04733636975288391, "sampling/sampling_logp_difference/max": 1.2221221923828125, "sampling/importance_sampling_ratio/min": 0.29460427165031433, "sampling/importance_sampling_ratio/mean": 1.0066455602645874, "sampling/importance_sampling_ratio/max": 1.8663012981414795, "entropy": 0.415604081004858, "clip_ratio/low_mean": 0.016570850741118193, "clip_ratio/low_min": 0.016570850741118193, "clip_ratio/high_mean": 0.023007763549685478, "clip_ratio/high_max": 0.023007763549685478, "clip_ratio/region_mean": 0.03957861429080367, "reward_total_mean": 0.6518079042434692, "reward_meter_mean": 0.9921485781669617, "reward_meter_std": 0.002238625893369317, "reward_count_adherence_mean": 0.8035714030265808, "reward_count_adherence_std": 0.07393559068441391, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8219696879386902, "reward_repeat_penalty_std": 0.08952570706605911, "reward_total_composite_mean": 0.6518079042434692, "reward_total_composite_std": 0.05643462389707565} {"timestamp_utc": "2026-04-12T00:16:30Z", "mode": "train", "global_step": 1483, "epoch": 0.059565409487086794, "loss": 0.0024, "grad_norm": 4.041107177734375, "learning_rate": 5.50909090909091e-06, "num_tokens": 3337913.0, "completions/mean_length": 159.75, "completions/min_length": 145.0, "completions/max_length": 164.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 159.75, "completions/min_terminated_length": 145.0, "completions/max_terminated_length": 164.0, "rewards/meter/mean": 0.8998898267745972, "rewards/meter/std": 0.24152173101902008, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.05892555043101311, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8854166865348816, "rewards/repeat_penalty/std": 0.10751881450414658, "rewards/total_composite/mean": 0.6401326656341553, "rewards/total_composite/std": 0.18946325778961182, "reward": 0.6401326656341553, "reward_std": 0.18946325778961182, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06710459291934967, "sampling/sampling_logp_difference/max": 2.618948221206665, "sampling/importance_sampling_ratio/min": 0.07287947088479996, "sampling/importance_sampling_ratio/mean": 1.0176057815551758, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6857388708740473, "clip_ratio/low_mean": 0.014218881260603666, "clip_ratio/low_min": 0.014218881260603666, "clip_ratio/high_mean": 0.03336897538974881, "clip_ratio/high_max": 0.03336897538974881, "clip_ratio/region_mean": 0.04758785665035248, "reward_total_mean": 0.6401326656341553, "reward_meter_mean": 0.8998898267745972, "reward_meter_std": 0.24152173101902008, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.05892555043101311, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8854166865348816, "reward_repeat_penalty_std": 0.10751881450414658, "reward_total_composite_mean": 0.6401326656341553, "reward_total_composite_std": 0.18946325778961182} {"timestamp_utc": "2026-04-12T00:16:34Z", "mode": "train", "global_step": 1484, "epoch": 0.05960557496887175, "loss": 0.0206, "grad_norm": 6.229678153991699, "learning_rate": 5.506060606060607e-06, "num_tokens": 3339388.0, "completions/mean_length": 33.375, "completions/min_length": 32.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.375, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9635177850723267, "rewards/meter/std": 0.020120054483413696, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9635177850723267, "rewards/total_composite/std": 0.020120054483413696, "reward": 0.9635177850723267, "reward_std": 0.02012004144489765, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.058272585272789, "sampling/sampling_logp_difference/max": 1.0730462074279785, "sampling/importance_sampling_ratio/min": 0.5093486905097961, "sampling/importance_sampling_ratio/mean": 1.0330264568328857, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5462967716157436, "clip_ratio/low_mean": 0.007464349502697587, "clip_ratio/low_min": 0.007464349502697587, "clip_ratio/high_mean": 0.018196672899648547, "clip_ratio/high_max": 0.018196672899648547, "clip_ratio/region_mean": 0.025661022402346134, "reward_total_mean": 0.9635177850723267, "reward_meter_mean": 0.9635177850723267, "reward_meter_std": 0.020120054483413696, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9635177850723267, "reward_total_composite_std": 0.020120054483413696} {"timestamp_utc": "2026-04-12T00:16:40Z", "mode": "train", "global_step": 1485, "epoch": 0.05964574045065671, "loss": 0.0024, "grad_norm": 4.295280456542969, "learning_rate": 5.5030303030303034e-06, "num_tokens": 3341484.0, "completions/mean_length": 90.0, "completions/min_length": 87.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.0, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.6956532597541809, "rewards/meter/std": 0.21607494354248047, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.6436759233474731, "rewards/total_composite/std": 0.27830082178115845, "reward": 0.6436759233474731, "reward_std": 0.27830079197883606, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07055231183767319, "sampling/sampling_logp_difference/max": 1.5324945449829102, "sampling/importance_sampling_ratio/min": 0.21599619090557098, "sampling/importance_sampling_ratio/mean": 1.0068004131317139, "sampling/importance_sampling_ratio/max": 1.6940116882324219, "entropy": 0.5103895664215088, "clip_ratio/low_mean": 0.006699939724057913, "clip_ratio/low_min": 0.006699939724057913, "clip_ratio/high_mean": 0.025170997832901776, "clip_ratio/high_max": 0.025170997832901776, "clip_ratio/region_mean": 0.03187093755695969, "reward_total_mean": 0.6436759233474731, "reward_meter_mean": 0.6956532597541809, "reward_meter_std": 0.21607494354248047, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.6436759233474731, "reward_total_composite_std": 0.27830082178115845} {"timestamp_utc": "2026-04-12T00:16:45Z", "mode": "train", "global_step": 1486, "epoch": 0.05968590593244166, "loss": 0.0174, "grad_norm": 5.7835164070129395, "learning_rate": 5.500000000000001e-06, "num_tokens": 3343427.0, "completions/mean_length": 85.875, "completions/min_length": 81.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.875, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.5018972158432007, "rewards/meter/std": 0.3999512195587158, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.41748714447021484, "rewards/total_composite/std": 0.3111114799976349, "reward": 0.41748714447021484, "reward_std": 0.3111114799976349, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0595453642308712, "sampling/sampling_logp_difference/max": 1.609797477722168, "sampling/importance_sampling_ratio/min": 0.19992808997631073, "sampling/importance_sampling_ratio/mean": 1.0213876962661743, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6251698546111584, "clip_ratio/low_mean": 0.021903468994423747, "clip_ratio/low_min": 0.021903468994423747, "clip_ratio/high_mean": 0.010355054982937872, "clip_ratio/high_max": 0.010355054982937872, "clip_ratio/region_mean": 0.03225852397736162, "reward_total_mean": 0.41748714447021484, "reward_meter_mean": 0.5018972158432007, "reward_meter_std": 0.3999512195587158, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_total_composite_mean": 0.41748714447021484, "reward_total_composite_std": 0.3111114799976349} {"timestamp_utc": "2026-04-12T00:16:50Z", "mode": "train", "global_step": 1487, "epoch": 0.05972607141422662, "loss": 0.0425, "grad_norm": 18.545215606689453, "learning_rate": 5.496969696969697e-06, "num_tokens": 3345165.0, "completions/mean_length": 55.25, "completions/min_length": 53.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.25, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9714435338973999, "rewards/meter/std": 0.04950811713933945, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9714435338973999, "rewards/total_composite/std": 0.04950811713933945, "reward": 0.9714435338973999, "reward_std": 0.04950810596346855, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06842397898435593, "sampling/sampling_logp_difference/max": 1.205331802368164, "sampling/importance_sampling_ratio/min": 0.2995925843715668, "sampling/importance_sampling_ratio/mean": 1.014945149421692, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.33666370064020157, "clip_ratio/low_mean": 0.0132503192871809, "clip_ratio/low_min": 0.0132503192871809, "clip_ratio/high_mean": 0.02728484314866364, "clip_ratio/high_max": 0.02728484314866364, "clip_ratio/region_mean": 0.04053516243584454, "reward_total_mean": 0.9714435338973999, "reward_meter_mean": 0.9714435338973999, "reward_meter_std": 0.04950811713933945, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9714435338973999, "reward_total_composite_std": 0.04950811713933945} {"timestamp_utc": "2026-04-12T00:16:55Z", "mode": "train", "global_step": 1488, "epoch": 0.05976623689601157, "loss": -0.0096, "grad_norm": 4.115540504455566, "learning_rate": 5.493939393939395e-06, "num_tokens": 3346850.0, "completions/mean_length": 55.625, "completions/min_length": 54.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.625, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.992435097694397, "rewards/meter/std": 0.006868826691061258, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.992435097694397, "rewards/total_composite/std": 0.006868826691061258, "reward": 0.992435097694397, "reward_std": 0.006868841592222452, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0397869236767292, "sampling/sampling_logp_difference/max": 1.0334584712982178, "sampling/importance_sampling_ratio/min": 0.3714584410190582, "sampling/importance_sampling_ratio/mean": 1.0044257640838623, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2505888659507036, "clip_ratio/low_mean": 0.0069444444961845875, "clip_ratio/low_min": 0.0069444444961845875, "clip_ratio/high_mean": 0.020023792050778866, "clip_ratio/high_max": 0.020023792050778866, "clip_ratio/region_mean": 0.026968236546963453, "reward_total_mean": 0.992435097694397, "reward_meter_mean": 0.992435097694397, "reward_meter_std": 0.006868826691061258, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.992435097694397, "reward_total_composite_std": 0.006868826691061258} {"timestamp_utc": "2026-04-12T00:17:01Z", "mode": "train", "global_step": 1489, "epoch": 0.059806402377796525, "loss": 0.0041, "grad_norm": 3.699056386947632, "learning_rate": 5.490909090909091e-06, "num_tokens": 3348897.0, "completions/mean_length": 90.875, "completions/min_length": 87.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.875, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.7553209066390991, "rewards/meter/std": 0.27557846903800964, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.7304862141609192, "rewards/total_composite/std": 0.2595451772212982, "reward": 0.7304862141609192, "reward_std": 0.2595451772212982, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07237864285707474, "sampling/sampling_logp_difference/max": 1.377182960510254, "sampling/importance_sampling_ratio/min": 0.252288281917572, "sampling/importance_sampling_ratio/mean": 1.0141518115997314, "sampling/importance_sampling_ratio/max": 1.844269871711731, "entropy": 0.6783212311565876, "clip_ratio/low_mean": 0.013681166106835008, "clip_ratio/low_min": 0.013681166106835008, "clip_ratio/high_mean": 0.04566098749637604, "clip_ratio/high_max": 0.04566098749637604, "clip_ratio/region_mean": 0.059342153603211045, "reward_total_mean": 0.7304862141609192, "reward_meter_mean": 0.7553209066390991, "reward_meter_std": 0.27557846903800964, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.7304862141609192, "reward_total_composite_std": 0.2595451772212982} {"timestamp_utc": "2026-04-12T00:17:07Z", "mode": "train", "global_step": 1490, "epoch": 0.05984656785958148, "loss": 0.019, "grad_norm": 3.8729865550994873, "learning_rate": 5.487878787878789e-06, "num_tokens": 3351734.0, "completions/mean_length": 155.625, "completions/min_length": 151.0, "completions/max_length": 160.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 155.625, "completions/min_terminated_length": 151.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.8863186240196228, "rewards/meter/std": 0.23323532938957214, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8888888955116272, "rewards/repeat_penalty/std": 0.10286889225244522, "rewards/total_composite/mean": 0.6489295363426208, "rewards/total_composite/std": 0.17390431463718414, "reward": 0.6489295363426208, "reward_std": 0.17390431463718414, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08673720061779022, "sampling/sampling_logp_difference/max": 1.3975944519042969, "sampling/importance_sampling_ratio/min": 0.24719087779521942, "sampling/importance_sampling_ratio/mean": 1.026315450668335, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8372621536254883, "clip_ratio/low_mean": 0.03902717446908355, "clip_ratio/low_min": 0.03902717446908355, "clip_ratio/high_mean": 0.03319170791655779, "clip_ratio/high_max": 0.03319170791655779, "clip_ratio/region_mean": 0.07221888238564134, "reward_total_mean": 0.6489295363426208, "reward_meter_mean": 0.8863186240196228, "reward_meter_std": 0.23323532938957214, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8888888955116272, "reward_repeat_penalty_std": 0.10286889225244522, "reward_total_composite_mean": 0.6489295363426208, "reward_total_composite_std": 0.17390431463718414} {"timestamp_utc": "2026-04-12T00:17:12Z", "mode": "train", "global_step": 1491, "epoch": 0.05988673334136643, "loss": -0.0074, "grad_norm": 5.608306884765625, "learning_rate": 5.484848484848485e-06, "num_tokens": 3353757.0, "completions/mean_length": 65.875, "completions/min_length": 64.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.875, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9181324243545532, "rewards/meter/std": 0.15554846823215485, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9181324243545532, "rewards/total_composite/std": 0.15554846823215485, "reward": 0.9181324243545532, "reward_std": 0.15554848313331604, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04674442857503891, "sampling/sampling_logp_difference/max": 1.3651247024536133, "sampling/importance_sampling_ratio/min": 0.2553488314151764, "sampling/importance_sampling_ratio/mean": 1.010719895362854, "sampling/importance_sampling_ratio/max": 1.6148149967193604, "entropy": 0.31265123561024666, "clip_ratio/low_mean": 0.013612689450383186, "clip_ratio/low_min": 0.013612689450383186, "clip_ratio/high_mean": 0.02461273316293955, "clip_ratio/high_max": 0.02461273316293955, "clip_ratio/region_mean": 0.038225422613322735, "reward_total_mean": 0.9181324243545532, "reward_meter_mean": 0.9181324243545532, "reward_meter_std": 0.15554846823215485, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9181324243545532, "reward_total_composite_std": 0.15554846823215485} {"timestamp_utc": "2026-04-12T00:17:16Z", "mode": "train", "global_step": 1492, "epoch": 0.05992689882315139, "loss": -0.0022, "grad_norm": 7.214973449707031, "learning_rate": 5.4818181818181825e-06, "num_tokens": 3355426.0, "completions/mean_length": 56.625, "completions/min_length": 53.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.625, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.7181947231292725, "rewards/meter/std": 0.3745432198047638, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7181947231292725, "rewards/total_composite/std": 0.3745432198047638, "reward": 0.7181947231292725, "reward_std": 0.3745432198047638, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07947485893964767, "sampling/sampling_logp_difference/max": 1.292196273803711, "sampling/importance_sampling_ratio/min": 0.2746668756008148, "sampling/importance_sampling_ratio/mean": 1.0328161716461182, "sampling/importance_sampling_ratio/max": 1.654712200164795, "entropy": 0.9116378463804722, "clip_ratio/low_mean": 0.025138093391433358, "clip_ratio/low_min": 0.025138093391433358, "clip_ratio/high_mean": 0.019893484190106392, "clip_ratio/high_max": 0.019893484190106392, "clip_ratio/region_mean": 0.04503157758153975, "reward_total_mean": 0.7181947231292725, "reward_meter_mean": 0.7181947231292725, "reward_meter_std": 0.3745432198047638, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7181947231292725, "reward_total_composite_std": 0.3745432198047638} {"timestamp_utc": "2026-04-12T00:17:23Z", "mode": "train", "global_step": 1493, "epoch": 0.05996706430493634, "loss": 0.0529, "grad_norm": 4.562529563903809, "learning_rate": 5.478787878787879e-06, "num_tokens": 3358873.0, "completions/mean_length": 235.875, "completions/min_length": 211.0, "completions/max_length": 259.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 235.875, "completions/min_terminated_length": 211.0, "completions/max_terminated_length": 259.0, "rewards/meter/mean": 0.9776490926742554, "rewards/meter/std": 0.024429455399513245, "rewards/count_adherence/mean": 0.7638888955116272, "rewards/count_adherence/std": 0.03928370773792267, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.8934294581413269, "rewards/repeat_penalty/std": 0.1227281242609024, "rewards/total_composite/mean": 0.5754737257957458, "rewards/total_composite/std": 0.2489534169435501, "reward": 0.5754737257957458, "reward_std": 0.2489534169435501, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07718866318464279, "sampling/sampling_logp_difference/max": 1.9858770370483398, "sampling/importance_sampling_ratio/min": 0.13726016879081726, "sampling/importance_sampling_ratio/mean": 1.0195549726486206, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8409713134169579, "clip_ratio/low_mean": 0.011852287454530597, "clip_ratio/low_min": 0.011852287454530597, "clip_ratio/high_mean": 0.03274800395593047, "clip_ratio/high_max": 0.03274800395593047, "clip_ratio/region_mean": 0.04460029141046107, "reward_total_mean": 0.5754737257957458, "reward_meter_mean": 0.9776490926742554, "reward_meter_std": 0.024429455399513245, "reward_count_adherence_mean": 0.7638888955116272, "reward_count_adherence_std": 0.03928370773792267, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.8934294581413269, "reward_repeat_penalty_std": 0.1227281242609024, "reward_total_composite_mean": 0.5754737257957458, "reward_total_composite_std": 0.2489534169435501} {"timestamp_utc": "2026-04-12T00:17:28Z", "mode": "train", "global_step": 1494, "epoch": 0.060007229786721294, "loss": 0.0076, "grad_norm": 7.282406806945801, "learning_rate": 5.475757575757576e-06, "num_tokens": 3360884.0, "completions/mean_length": 85.375, "completions/min_length": 78.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.375, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.665711522102356, "rewards/meter/std": 0.4568917751312256, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.6164090037345886, "rewards/total_composite/std": 0.42542141675949097, "reward": 0.6164090037345886, "reward_std": 0.4254213869571686, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09246491640806198, "sampling/sampling_logp_difference/max": 1.6489763259887695, "sampling/importance_sampling_ratio/min": 0.19224660098552704, "sampling/importance_sampling_ratio/mean": 1.025902509689331, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0446771308779716, "clip_ratio/low_mean": 0.028380894218571484, "clip_ratio/low_min": 0.028380894218571484, "clip_ratio/high_mean": 0.014402368804439902, "clip_ratio/high_max": 0.014402368804439902, "clip_ratio/region_mean": 0.042783263023011386, "reward_total_mean": 0.6164090037345886, "reward_meter_mean": 0.665711522102356, "reward_meter_std": 0.4568917751312256, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.6164090037345886, "reward_total_composite_std": 0.42542141675949097} {"timestamp_utc": "2026-04-12T00:17:33Z", "mode": "train", "global_step": 1495, "epoch": 0.06004739526850625, "loss": 0.0083, "grad_norm": 3.7021563053131104, "learning_rate": 5.472727272727273e-06, "num_tokens": 3362761.0, "completions/mean_length": 65.625, "completions/min_length": 64.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.625, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9978386163711548, "rewards/meter/std": 0.0011219180887565017, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978386163711548, "rewards/total_composite/std": 0.0011219180887565017, "reward": 0.9978386163711548, "reward_std": 0.0011219230946153402, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019154906272888184, "sampling/sampling_logp_difference/max": 0.6693814992904663, "sampling/importance_sampling_ratio/min": 0.5120251774787903, "sampling/importance_sampling_ratio/mean": 1.000672698020935, "sampling/importance_sampling_ratio/max": 1.4048590660095215, "entropy": 0.10151981003582478, "clip_ratio/low_mean": 0.007462686393409967, "clip_ratio/low_min": 0.007462686393409967, "clip_ratio/high_mean": 0.011480186600238085, "clip_ratio/high_max": 0.011480186600238085, "clip_ratio/region_mean": 0.018942872993648052, "reward_total_mean": 0.9978386163711548, "reward_meter_mean": 0.9978386163711548, "reward_meter_std": 0.0011219180887565017, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9978386163711548, "reward_total_composite_std": 0.0011219180887565017} {"timestamp_utc": "2026-04-12T00:17:38Z", "mode": "train", "global_step": 1496, "epoch": 0.0600875607502912, "loss": 0.0026, "grad_norm": 4.584413528442383, "learning_rate": 5.469696969696971e-06, "num_tokens": 3364575.0, "completions/mean_length": 65.75, "completions/min_length": 63.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.75, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9560348987579346, "rewards/meter/std": 0.09728128463029861, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9560348987579346, "rewards/total_composite/std": 0.09728128463029861, "reward": 0.9560348987579346, "reward_std": 0.09728129208087921, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04986903816461563, "sampling/sampling_logp_difference/max": 1.1417694091796875, "sampling/importance_sampling_ratio/min": 0.3192536234855652, "sampling/importance_sampling_ratio/mean": 1.0161854028701782, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3818973954766989, "clip_ratio/low_mean": 0.001923076924867928, "clip_ratio/low_min": 0.001923076924867928, "clip_ratio/high_mean": 0.02847922092769295, "clip_ratio/high_max": 0.02847922092769295, "clip_ratio/region_mean": 0.030402297852560878, "reward_total_mean": 0.9560348987579346, "reward_meter_mean": 0.9560348987579346, "reward_meter_std": 0.09728128463029861, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9560348987579346, "reward_total_composite_std": 0.09728128463029861} {"timestamp_utc": "2026-04-12T00:17:43Z", "mode": "train", "global_step": 1497, "epoch": 0.060127726232076156, "loss": 0.0391, "grad_norm": 6.656605243682861, "learning_rate": 5.466666666666667e-06, "num_tokens": 3366568.0, "completions/mean_length": 89.125, "completions/min_length": 84.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.125, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.6094886064529419, "rewards/meter/std": 0.4335397183895111, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.5929021239280701, "rewards/total_composite/std": 0.4219600558280945, "reward": 0.5929021239280701, "reward_std": 0.4219600260257721, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08898783475160599, "sampling/sampling_logp_difference/max": 1.6831512451171875, "sampling/importance_sampling_ratio/min": 0.18578758835792542, "sampling/importance_sampling_ratio/mean": 1.0115280151367188, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7317456901073456, "clip_ratio/low_mean": 0.01920653972774744, "clip_ratio/low_min": 0.01920653972774744, "clip_ratio/high_mean": 0.05977132357656956, "clip_ratio/high_max": 0.05977132357656956, "clip_ratio/region_mean": 0.078977863304317, "reward_total_mean": 0.5929021239280701, "reward_meter_mean": 0.6094886064529419, "reward_meter_std": 0.4335397183895111, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.5929021239280701, "reward_total_composite_std": 0.4219600558280945} {"timestamp_utc": "2026-04-12T00:17:47Z", "mode": "train", "global_step": 1498, "epoch": 0.06016789171386111, "loss": 0.0331, "grad_norm": 9.448737144470215, "learning_rate": 5.463636363636364e-06, "num_tokens": 3368045.0, "completions/mean_length": 34.625, "completions/min_length": 33.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.625, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9881667494773865, "rewards/meter/std": 0.003932104911655188, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9881667494773865, "rewards/total_composite/std": 0.003932104911655188, "reward": 0.9881667494773865, "reward_std": 0.003932103049010038, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04901129752397537, "sampling/sampling_logp_difference/max": 1.6449816226959229, "sampling/importance_sampling_ratio/min": 0.19301611185073853, "sampling/importance_sampling_ratio/mean": 1.021955132484436, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3656330443918705, "clip_ratio/low_mean": 0.03597753681242466, "clip_ratio/low_min": 0.03597753681242466, "clip_ratio/high_mean": 0.017676767893135548, "clip_ratio/high_max": 0.017676767893135548, "clip_ratio/region_mean": 0.05365430470556021, "reward_total_mean": 0.9881667494773865, "reward_meter_mean": 0.9881667494773865, "reward_meter_std": 0.003932104911655188, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9881667494773865, "reward_total_composite_std": 0.003932104911655188} {"timestamp_utc": "2026-04-12T00:17:53Z", "mode": "train", "global_step": 1499, "epoch": 0.060208057195646064, "loss": 0.0209, "grad_norm": 3.978285551071167, "learning_rate": 5.460606060606061e-06, "num_tokens": 3370801.0, "completions/mean_length": 128.5, "completions/min_length": 120.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.5, "completions/min_terminated_length": 120.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.9624004364013672, "rewards/meter/std": 0.06776177138090134, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.9302494525909424, "rewards/total_composite/std": 0.10968229919672012, "reward": 0.9302494525909424, "reward_std": 0.10968230664730072, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09097252786159515, "sampling/sampling_logp_difference/max": 1.2857685089111328, "sampling/importance_sampling_ratio/min": 0.27643805742263794, "sampling/importance_sampling_ratio/mean": 1.029836893081665, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9432215169072151, "clip_ratio/low_mean": 0.024429237004369497, "clip_ratio/low_min": 0.024429237004369497, "clip_ratio/high_mean": 0.05129538010805845, "clip_ratio/high_max": 0.05129538010805845, "clip_ratio/region_mean": 0.07572461711242795, "reward_total_mean": 0.9302494525909424, "reward_meter_mean": 0.9624004364013672, "reward_meter_std": 0.06776177138090134, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.9302494525909424, "reward_total_composite_std": 0.10968229919672012} {"timestamp_utc": "2026-04-12T00:17:57Z", "mode": "train", "global_step": 1500, "epoch": 0.06024822267743102, "loss": -0.0122, "grad_norm": 5.225851058959961, "learning_rate": 5.457575757575758e-06, "num_tokens": 3372511.0, "completions/mean_length": 63.75, "completions/min_length": 60.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.75, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9959254264831543, "rewards/meter/std": 0.0024275805335491896, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9959254264831543, "rewards/total_composite/std": 0.0024275805335491896, "reward": 0.9959254264831543, "reward_std": 0.002427596366032958, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07604724913835526, "sampling/sampling_logp_difference/max": 0.9057703018188477, "sampling/importance_sampling_ratio/min": 0.40423041582107544, "sampling/importance_sampling_ratio/mean": 1.0106977224349976, "sampling/importance_sampling_ratio/max": 1.7771986722946167, "entropy": 0.7506604716181755, "clip_ratio/low_mean": 0.02421415294520557, "clip_ratio/low_min": 0.02421415294520557, "clip_ratio/high_mean": 0.01567198208067566, "clip_ratio/high_max": 0.01567198208067566, "clip_ratio/region_mean": 0.03988613502588123, "reward_total_mean": 0.9959254264831543, "reward_meter_mean": 0.9959254264831543, "reward_meter_std": 0.0024275805335491896, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9959254264831543, "reward_total_composite_std": 0.0024275805335491896} {"timestamp_utc": "2026-04-12T00:18:57Z", "mode": "eval", "global_step": 1500, "epoch": 0.06024822267743102, "eval_loss": NaN, "eval_runtime": 59.2586, "eval_samples_per_second": 1.755, "eval_steps_per_second": 0.219, "eval_num_tokens": 3372511.0, "eval_completions/mean_length": 163.32692307692307, "eval_completions/min_length": 57.92307692307692, "eval_completions/max_length": 304.38461538461536, "eval_completions/clipped_ratio": 0.019230769230769232, "eval_completions/mean_terminated_length": 156.44917766864484, "eval_completions/min_terminated_length": 57.92307692307692, "eval_completions/max_terminated_length": 271.9230769230769, "eval_rewards/meter/mean": 0.6285286339429709, "eval_rewards/meter/std": 0.44250402083763707, "eval_rewards/count_adherence/mean": 0.8234805831542382, "eval_rewards/count_adherence/std": 0.17093561990902975, "eval_rewards/arabic_clean/mean": 0.9711538461538461, "eval_rewards/arabic_clean/std": 0.08158924258672275, "eval_rewards/repeat_penalty/mean": 0.8920627649013813, "eval_rewards/repeat_penalty/std": 0.12081348953338769, "eval_rewards/total_composite/mean": 0.4787046152811784, "eval_rewards/total_composite/std": 0.3746733757165762, "eval_reward": 0.4787046152811784, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.04665847968023557, "eval_sampling/sampling_logp_difference/max": 1.12705293068519, "eval_sampling/importance_sampling_ratio/min": 0.32900666961303127, "eval_sampling/importance_sampling_ratio/mean": 1.0148254266152015, "eval_sampling/importance_sampling_ratio/max": 1.5625533782518828, "eval_entropy": 0.5243608149198385, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.4787046152811784, "eval_reward_meter_mean": 0.6285286339429709, "eval_reward_meter_std": 0.44250402083763707, "eval_reward_count_adherence_mean": 0.8234805831542382, "eval_reward_count_adherence_std": 0.17093561990902975, "eval_reward_arabic_clean_mean": 0.9711538461538461, "eval_reward_arabic_clean_std": 0.08158924258672275, "eval_reward_repeat_penalty_mean": 0.8920627649013813, "eval_reward_repeat_penalty_std": 0.12081348953338769, "eval_reward_total_composite_mean": 0.4787046152811784, "eval_reward_total_composite_std": 0.3746733757165762} {"timestamp_utc": "2026-04-12T00:19:04Z", "mode": "train", "global_step": 1501, "epoch": 0.06028838815921597, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.4545454545454545e-06, "num_tokens": 3374031.0, "completions/mean_length": 33.0, "completions/min_length": 33.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9988001585006714, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9988001585006714, "rewards/total_composite/std": 0.0, "reward": 0.9988001585006714, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0034284384455531836, "sampling/sampling_logp_difference/max": 0.05314040184020996, "sampling/importance_sampling_ratio/min": 0.9900038242340088, "sampling/importance_sampling_ratio/mean": 1.0031137466430664, "sampling/importance_sampling_ratio/max": 1.0545777082443237, "entropy": 0.031635227147489786, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9988001585006714, "reward_meter_mean": 0.9988001585006714, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9988001585006714, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T00:19:09Z", "mode": "train", "global_step": 1502, "epoch": 0.060328553641000926, "loss": 0.0069, "grad_norm": 5.628778457641602, "learning_rate": 5.451515151515152e-06, "num_tokens": 3376224.0, "completions/mean_length": 98.125, "completions/min_length": 93.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.125, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.8838797807693481, "rewards/meter/std": 0.2619064450263977, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8838797807693481, "rewards/total_composite/std": 0.2619064450263977, "reward": 0.8838797807693481, "reward_std": 0.2619064748287201, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06008461117744446, "sampling/sampling_logp_difference/max": 1.558868408203125, "sampling/importance_sampling_ratio/min": 0.21037398278713226, "sampling/importance_sampling_ratio/mean": 1.0124897956848145, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5762354806065559, "clip_ratio/low_mean": 0.020408162847161293, "clip_ratio/low_min": 0.020408162847161293, "clip_ratio/high_mean": 0.028100816532969475, "clip_ratio/high_max": 0.028100816532969475, "clip_ratio/region_mean": 0.04850897938013077, "reward_total_mean": 0.8838797807693481, "reward_meter_mean": 0.8838797807693481, "reward_meter_std": 0.2619064450263977, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8838797807693481, "reward_total_composite_std": 0.2619064450263977} {"timestamp_utc": "2026-04-12T00:19:14Z", "mode": "train", "global_step": 1503, "epoch": 0.06036871912278588, "loss": -0.0001, "grad_norm": 2.2546792030334473, "learning_rate": 5.448484848484848e-06, "num_tokens": 3378172.0, "completions/mean_length": 65.5, "completions/min_length": 63.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.5, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.998626708984375, "rewards/meter/std": 0.00024546196800656617, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998626708984375, "rewards/total_composite/std": 0.00024546196800656617, "reward": 0.998626708984375, "reward_std": 0.0002454565546941012, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021490152925252914, "sampling/sampling_logp_difference/max": 1.5800909996032715, "sampling/importance_sampling_ratio/min": 0.2059563547372818, "sampling/importance_sampling_ratio/mean": 1.001434326171875, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10324481781572104, "clip_ratio/low_mean": 0.009538664249703288, "clip_ratio/low_min": 0.009538664249703288, "clip_ratio/high_mean": 0.001923076924867928, "clip_ratio/high_max": 0.001923076924867928, "clip_ratio/region_mean": 0.011461741174571216, "reward_total_mean": 0.998626708984375, "reward_meter_mean": 0.998626708984375, "reward_meter_std": 0.00024546196800656617, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.998626708984375, "reward_total_composite_std": 0.00024546196800656617} {"timestamp_utc": "2026-04-12T00:19:23Z", "mode": "train", "global_step": 1504, "epoch": 0.060408884604570834, "loss": -0.0139, "grad_norm": 2.0835001468658447, "learning_rate": 5.445454545454546e-06, "num_tokens": 3383481.0, "completions/mean_length": 384.625, "completions/min_length": 366.0, "completions/max_length": 408.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 384.625, "completions/min_terminated_length": 366.0, "completions/max_terminated_length": 408.0, "rewards/meter/mean": 0.9722083806991577, "rewards/meter/std": 0.05760227516293526, "rewards/count_adherence/mean": 0.6041666269302368, "rewards/count_adherence/std": 0.01964184269309044, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8138812780380249, "rewards/repeat_penalty/std": 0.13613860309123993, "rewards/total_composite/mean": 0.4811665415763855, "rewards/total_composite/std": 0.10122059285640717, "reward": 0.4811665415763855, "reward_std": 0.10122059285640717, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04671372473239899, "sampling/sampling_logp_difference/max": 1.6484105587005615, "sampling/importance_sampling_ratio/min": 0.19235540926456451, "sampling/importance_sampling_ratio/mean": 1.0164600610733032, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.44889700040221214, "clip_ratio/low_mean": 0.008970056660473347, "clip_ratio/low_min": 0.008970056660473347, "clip_ratio/high_mean": 0.019740989664569497, "clip_ratio/high_max": 0.019740989664569497, "clip_ratio/region_mean": 0.028711046325042844, "reward_total_mean": 0.4811665415763855, "reward_meter_mean": 0.9722083806991577, "reward_meter_std": 0.05760227516293526, "reward_count_adherence_mean": 0.6041666269302368, "reward_count_adherence_std": 0.01964184269309044, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8138812780380249, "reward_repeat_penalty_std": 0.13613860309123993, "reward_total_composite_mean": 0.4811665415763855, "reward_total_composite_std": 0.10122059285640717} {"timestamp_utc": "2026-04-12T00:19:30Z", "mode": "train", "global_step": 1505, "epoch": 0.06044905008635579, "loss": -0.0283, "grad_norm": 2.243826389312744, "learning_rate": 5.442424242424243e-06, "num_tokens": 3387080.0, "completions/mean_length": 249.875, "completions/min_length": 232.0, "completions/max_length": 267.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 249.875, "completions/min_terminated_length": 232.0, "completions/max_terminated_length": 267.0, "rewards/meter/mean": 0.9895928502082825, "rewards/meter/std": 0.008481908589601517, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0534522607922554, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.8847527503967285, "rewards/repeat_penalty/std": 0.05292898043990135, "rewards/total_composite/mean": 0.584118127822876, "rewards/total_composite/std": 0.24487483501434326, "reward": 0.584118127822876, "reward_std": 0.24487483501434326, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.045951321721076965, "sampling/sampling_logp_difference/max": 1.492568016052246, "sampling/importance_sampling_ratio/min": 0.2247946411371231, "sampling/importance_sampling_ratio/mean": 1.0099492073059082, "sampling/importance_sampling_ratio/max": 1.8827742338180542, "entropy": 0.4347769767045975, "clip_ratio/low_mean": 0.006312201963737607, "clip_ratio/low_min": 0.006312201963737607, "clip_ratio/high_mean": 0.035130204167217016, "clip_ratio/high_max": 0.035130204167217016, "clip_ratio/region_mean": 0.04144240613095462, "reward_total_mean": 0.584118127822876, "reward_meter_mean": 0.9895928502082825, "reward_meter_std": 0.008481908589601517, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0534522607922554, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.8847527503967285, "reward_repeat_penalty_std": 0.05292898043990135, "reward_total_composite_mean": 0.584118127822876, "reward_total_composite_std": 0.24487483501434326} {"timestamp_utc": "2026-04-12T00:19:35Z", "mode": "train", "global_step": 1506, "epoch": 0.06048921556814074, "loss": 0.0015, "grad_norm": 2.0182154178619385, "learning_rate": 5.43939393939394e-06, "num_tokens": 3388903.0, "completions/mean_length": 65.875, "completions/min_length": 65.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.875, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9988133907318115, "rewards/meter/std": 0.00020309248066041619, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9988133907318115, "rewards/total_composite/std": 0.00020309248066041619, "reward": 0.9988133907318115, "reward_std": 0.00020310519903432578, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009212362580001354, "sampling/sampling_logp_difference/max": 0.6020421981811523, "sampling/importance_sampling_ratio/min": 0.5476920008659363, "sampling/importance_sampling_ratio/mean": 1.0014187097549438, "sampling/importance_sampling_ratio/max": 1.8056702613830566, "entropy": 0.06401140103116632, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/high_mean": 0.005710955825634301, "clip_ratio/high_max": 0.005710955825634301, "clip_ratio/region_mean": 0.007604895276017487, "reward_total_mean": 0.9988133907318115, "reward_meter_mean": 0.9988133907318115, "reward_meter_std": 0.00020309248066041619, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9988133907318115, "reward_total_composite_std": 0.00020309248066041619} {"timestamp_utc": "2026-04-12T00:19:41Z", "mode": "train", "global_step": 1507, "epoch": 0.060529381049925696, "loss": 0.0083, "grad_norm": 3.6379973888397217, "learning_rate": 5.436363636363636e-06, "num_tokens": 3391499.0, "completions/mean_length": 125.5, "completions/min_length": 123.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 125.5, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9976436495780945, "rewards/meter/std": 0.0009433169034309685, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9441857933998108, "rewards/total_composite/std": 0.07358824461698532, "reward": 0.9441857933998108, "reward_std": 0.07358825951814651, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07620169967412949, "sampling/sampling_logp_difference/max": 1.429516315460205, "sampling/importance_sampling_ratio/min": 0.2394247055053711, "sampling/importance_sampling_ratio/mean": 1.0165196657180786, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6917183268815279, "clip_ratio/low_mean": 0.006993599818088114, "clip_ratio/low_min": 0.006993599818088114, "clip_ratio/high_mean": 0.04194163717329502, "clip_ratio/high_max": 0.04194163717329502, "clip_ratio/region_mean": 0.048935236991383135, "reward_total_mean": 0.9441857933998108, "reward_meter_mean": 0.9976436495780945, "reward_meter_std": 0.0009433169034309685, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.9441857933998108, "reward_total_composite_std": 0.07358824461698532} {"timestamp_utc": "2026-04-12T00:19:48Z", "mode": "train", "global_step": 1508, "epoch": 0.06056954653171065, "loss": 0.0189, "grad_norm": 5.184098243713379, "learning_rate": 5.4333333333333335e-06, "num_tokens": 3393400.0, "completions/mean_length": 90.625, "completions/min_length": 86.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.625, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.9280390739440918, "rewards/meter/std": 0.12549975514411926, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9280390739440918, "rewards/total_composite/std": 0.12549975514411926, "reward": 0.9280390739440918, "reward_std": 0.12549975514411926, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11991255730390549, "sampling/sampling_logp_difference/max": 1.5914983749389648, "sampling/importance_sampling_ratio/min": 0.2036202847957611, "sampling/importance_sampling_ratio/mean": 1.030521035194397, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3208992555737495, "clip_ratio/low_mean": 0.018996416125446558, "clip_ratio/low_min": 0.018996416125446558, "clip_ratio/high_mean": 0.050818526186048985, "clip_ratio/high_max": 0.050818526186048985, "clip_ratio/region_mean": 0.06981494231149554, "reward_total_mean": 0.9280390739440918, "reward_meter_mean": 0.9280390739440918, "reward_meter_std": 0.12549975514411926, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9280390739440918, "reward_total_composite_std": 0.12549975514411926} {"timestamp_utc": "2026-04-12T00:19:53Z", "mode": "train", "global_step": 1509, "epoch": 0.060609712013495604, "loss": -0.0186, "grad_norm": 5.793516159057617, "learning_rate": 5.430303030303032e-06, "num_tokens": 3395804.0, "completions/mean_length": 113.5, "completions/min_length": 104.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 113.5, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.9283153414726257, "rewards/meter/std": 0.17925596237182617, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.1079898476600647, "rewards/total_composite/mean": 0.8047507405281067, "rewards/total_composite/std": 0.20624437928199768, "reward": 0.8047507405281067, "reward_std": 0.20624437928199768, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03325022757053375, "sampling/sampling_logp_difference/max": 1.20695161819458, "sampling/importance_sampling_ratio/min": 0.29910770058631897, "sampling/importance_sampling_ratio/mean": 1.0089442729949951, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.22751092724502087, "clip_ratio/low_mean": 0.008413461968302727, "clip_ratio/low_min": 0.008413461968302727, "clip_ratio/high_mean": 0.015109836007468402, "clip_ratio/high_max": 0.015109836007468402, "clip_ratio/region_mean": 0.02352329797577113, "reward_total_mean": 0.8047507405281067, "reward_meter_mean": 0.9283153414726257, "reward_meter_std": 0.17925596237182617, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.1079898476600647, "reward_total_composite_mean": 0.8047507405281067, "reward_total_composite_std": 0.20624437928199768} {"timestamp_utc": "2026-04-12T00:19:58Z", "mode": "train", "global_step": 1510, "epoch": 0.06064987749528056, "loss": 0.0003, "grad_norm": 6.951155662536621, "learning_rate": 5.427272727272728e-06, "num_tokens": 3397499.0, "completions/mean_length": 57.875, "completions/min_length": 55.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.875, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.7507508397102356, "rewards/meter/std": 0.3587440252304077, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7507508397102356, "rewards/total_composite/std": 0.3587440252304077, "reward": 0.7507508397102356, "reward_std": 0.3587440550327301, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.053551964461803436, "sampling/sampling_logp_difference/max": 1.5763216018676758, "sampling/importance_sampling_ratio/min": 0.20673415064811707, "sampling/importance_sampling_ratio/mean": 1.015701174736023, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4132865481078625, "clip_ratio/low_mean": 0.010469448054209352, "clip_ratio/low_min": 0.010469448054209352, "clip_ratio/high_mean": 0.023895947029814124, "clip_ratio/high_max": 0.023895947029814124, "clip_ratio/region_mean": 0.034365395084023476, "reward_total_mean": 0.7507508397102356, "reward_meter_mean": 0.7507508397102356, "reward_meter_std": 0.3587440252304077, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7507508397102356, "reward_total_composite_std": 0.3587440252304077} {"timestamp_utc": "2026-04-12T00:20:07Z", "mode": "train", "global_step": 1511, "epoch": 0.06069004297706551, "loss": -0.0464, "grad_norm": 3.092373847961426, "learning_rate": 5.424242424242425e-06, "num_tokens": 3399092.0, "completions/mean_length": 107.125, "completions/min_length": 46.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 49.28571701049805, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.42876946926116943, "rewards/meter/std": 0.4184267520904541, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.42717820405960083, "rewards/total_composite/std": 0.4202551543712616, "reward": 0.42717820405960083, "reward_std": 0.420255184173584, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07508043199777603, "sampling/sampling_logp_difference/max": 1.4143075942993164, "sampling/importance_sampling_ratio/min": 0.24309387803077698, "sampling/importance_sampling_ratio/mean": 1.0123780965805054, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5180366039276123, "clip_ratio/low_mean": 0.02230392163619399, "clip_ratio/low_min": 0.02230392163619399, "clip_ratio/high_mean": 0.022818502504378557, "clip_ratio/high_max": 0.022818502504378557, "clip_ratio/region_mean": 0.04512242414057255, "reward_total_mean": 0.42717820405960083, "reward_meter_mean": 0.42876946926116943, "reward_meter_std": 0.4184267520904541, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.42717820405960083, "reward_total_composite_std": 0.4202551543712616} {"timestamp_utc": "2026-04-12T00:20:17Z", "mode": "train", "global_step": 1512, "epoch": 0.060730208458850465, "loss": -0.3845, "grad_norm": 1.4069581031799316, "learning_rate": 5.421212121212122e-06, "num_tokens": 3402818.0, "completions/mean_length": 405.75, "completions/min_length": 309.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 342.0, "completions/min_terminated_length": 309.0, "completions/max_terminated_length": 367.0, "rewards/meter/mean": 0.7066450119018555, "rewards/meter/std": 0.3806198835372925, "rewards/count_adherence/mean": 0.6102941036224365, "rewards/count_adherence/std": 0.08858475089073181, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/repeat_penalty/mean": 0.9273183345794678, "rewards/repeat_penalty/std": 0.11418548226356506, "rewards/total_composite/mean": 0.33205440640449524, "rewards/total_composite/std": 0.27961450815200806, "reward": 0.33205440640449524, "reward_std": 0.27961447834968567, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08264101296663284, "sampling/sampling_logp_difference/max": 1.9105606079101562, "sampling/importance_sampling_ratio/min": 0.1479973942041397, "sampling/importance_sampling_ratio/mean": 1.020935297012329, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5392166823148727, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.031007058219984174, "clip_ratio/high_max": 0.031007058219984174, "clip_ratio/region_mean": 0.031007058219984174, "reward_total_mean": 0.33205440640449524, "reward_meter_mean": 0.7066450119018555, "reward_meter_std": 0.3806198835372925, "reward_count_adherence_mean": 0.6102941036224365, "reward_count_adherence_std": 0.08858475089073181, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_repeat_penalty_mean": 0.9273183345794678, "reward_repeat_penalty_std": 0.11418548226356506, "reward_total_composite_mean": 0.33205440640449524, "reward_total_composite_std": 0.27961450815200806} {"timestamp_utc": "2026-04-12T00:20:22Z", "mode": "train", "global_step": 1513, "epoch": 0.06077037394063542, "loss": 0.0171, "grad_norm": 4.46673583984375, "learning_rate": 5.418181818181819e-06, "num_tokens": 3404619.0, "completions/mean_length": 67.125, "completions/min_length": 66.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.125, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9849165081977844, "rewards/meter/std": 0.019791634753346443, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9849165081977844, "rewards/total_composite/std": 0.019791634753346443, "reward": 0.9849165081977844, "reward_std": 0.019791632890701294, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025447804480791092, "sampling/sampling_logp_difference/max": 0.6372613906860352, "sampling/importance_sampling_ratio/min": 0.5287384390830994, "sampling/importance_sampling_ratio/mean": 1.0061801671981812, "sampling/importance_sampling_ratio/max": 1.5401774644851685, "entropy": 0.2136769648641348, "clip_ratio/low_mean": 0.005488860071636736, "clip_ratio/low_min": 0.005488860071636736, "clip_ratio/high_mean": 0.014931121026165783, "clip_ratio/high_max": 0.014931121026165783, "clip_ratio/region_mean": 0.02041998109780252, "reward_total_mean": 0.9849165081977844, "reward_meter_mean": 0.9849165081977844, "reward_meter_std": 0.019791634753346443, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9849165081977844, "reward_total_composite_std": 0.019791634753346443} {"timestamp_utc": "2026-04-12T00:20:28Z", "mode": "train", "global_step": 1514, "epoch": 0.06081053942242037, "loss": -0.0047, "grad_norm": 2.8653032779693604, "learning_rate": 5.415151515151515e-06, "num_tokens": 3408083.0, "completions/mean_length": 228.0, "completions/min_length": 218.0, "completions/max_length": 236.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 228.0, "completions/min_terminated_length": 218.0, "completions/max_terminated_length": 236.0, "rewards/meter/mean": 0.9303399324417114, "rewards/meter/std": 0.1539093255996704, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9519230723381042, "rewards/repeat_penalty/std": 0.05723259598016739, "rewards/total_composite/mean": 0.7759657502174377, "rewards/total_composite/std": 0.14068926870822906, "reward": 0.7759657502174377, "reward_std": 0.14068925380706787, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08908653259277344, "sampling/sampling_logp_difference/max": 1.7966899871826172, "sampling/importance_sampling_ratio/min": 0.16584692895412445, "sampling/importance_sampling_ratio/mean": 1.0278751850128174, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0589845478534698, "clip_ratio/low_mean": 0.012164860963821411, "clip_ratio/low_min": 0.012164860963821411, "clip_ratio/high_mean": 0.04533929843455553, "clip_ratio/high_max": 0.04533929843455553, "clip_ratio/region_mean": 0.05750415939837694, "reward_total_mean": 0.7759657502174377, "reward_meter_mean": 0.9303399324417114, "reward_meter_std": 0.1539093255996704, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9519230723381042, "reward_repeat_penalty_std": 0.05723259598016739, "reward_total_composite_mean": 0.7759657502174377, "reward_total_composite_std": 0.14068926870822906} {"timestamp_utc": "2026-04-12T00:20:33Z", "mode": "train", "global_step": 1515, "epoch": 0.06085070490420533, "loss": -0.0027, "grad_norm": 3.14272403717041, "learning_rate": 5.412121212121213e-06, "num_tokens": 3410076.0, "completions/mean_length": 68.125, "completions/min_length": 67.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.125, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9943439960479736, "rewards/meter/std": 0.0008108045440167189, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9943439960479736, "rewards/total_composite/std": 0.0008108045440167189, "reward": 0.9943439960479736, "reward_std": 0.0008108008187264204, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03039509244263172, "sampling/sampling_logp_difference/max": 0.9262442588806152, "sampling/importance_sampling_ratio/min": 0.39603835344314575, "sampling/importance_sampling_ratio/mean": 1.0022821426391602, "sampling/importance_sampling_ratio/max": 1.5258408784866333, "entropy": 0.2124041486531496, "clip_ratio/low_mean": 0.0055147059028968215, "clip_ratio/low_min": 0.0055147059028968215, "clip_ratio/high_mean": 0.0221144916722551, "clip_ratio/high_max": 0.0221144916722551, "clip_ratio/region_mean": 0.02762919757515192, "reward_total_mean": 0.9943439960479736, "reward_meter_mean": 0.9943439960479736, "reward_meter_std": 0.0008108045440167189, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9943439960479736, "reward_total_composite_std": 0.0008108045440167189} {"timestamp_utc": "2026-04-12T00:20:38Z", "mode": "train", "global_step": 1516, "epoch": 0.06089087038599028, "loss": -0.0086, "grad_norm": 6.165884017944336, "learning_rate": 5.409090909090909e-06, "num_tokens": 3412012.0, "completions/mean_length": 67.0, "completions/min_length": 64.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9924481511116028, "rewards/meter/std": 0.005331422667950392, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9924481511116028, "rewards/total_composite/std": 0.005331422667950392, "reward": 0.9924481511116028, "reward_std": 0.0053314100950956345, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028799764811992645, "sampling/sampling_logp_difference/max": 1.173213005065918, "sampling/importance_sampling_ratio/min": 0.30937132239341736, "sampling/importance_sampling_ratio/mean": 1.0072860717773438, "sampling/importance_sampling_ratio/max": 1.659652829170227, "entropy": 0.19451193884015083, "clip_ratio/low_mean": 0.003818796598352492, "clip_ratio/low_min": 0.003818796598352492, "clip_ratio/high_mean": 0.009246049099601805, "clip_ratio/high_max": 0.009246049099601805, "clip_ratio/region_mean": 0.013064845697954297, "reward_total_mean": 0.9924481511116028, "reward_meter_mean": 0.9924481511116028, "reward_meter_std": 0.005331422667950392, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9924481511116028, "reward_total_composite_std": 0.005331422667950392} {"timestamp_utc": "2026-04-12T00:20:42Z", "mode": "train", "global_step": 1517, "epoch": 0.060931035867775235, "loss": -0.015, "grad_norm": 2.4998230934143066, "learning_rate": 5.406060606060607e-06, "num_tokens": 3413722.0, "completions/mean_length": 57.75, "completions/min_length": 55.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.75, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9230245351791382, "rewards/meter/std": 0.1628594845533371, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9230245351791382, "rewards/total_composite/std": 0.1628594845533371, "reward": 0.9230245351791382, "reward_std": 0.1628594696521759, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04387671872973442, "sampling/sampling_logp_difference/max": 1.2724885940551758, "sampling/importance_sampling_ratio/min": 0.28013360500335693, "sampling/importance_sampling_ratio/mean": 1.0011502504348755, "sampling/importance_sampling_ratio/max": 1.4794626235961914, "entropy": 0.31292806193232536, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.027737777214497328, "clip_ratio/high_max": 0.027737777214497328, "clip_ratio/region_mean": 0.027737777214497328, "reward_total_mean": 0.9230245351791382, "reward_meter_mean": 0.9230245351791382, "reward_meter_std": 0.1628594845533371, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9230245351791382, "reward_total_composite_std": 0.1628594845533371} {"timestamp_utc": "2026-04-12T00:20:47Z", "mode": "train", "global_step": 1518, "epoch": 0.06097120134956019, "loss": 0.0093, "grad_norm": 6.170195579528809, "learning_rate": 5.4030303030303036e-06, "num_tokens": 3415375.0, "completions/mean_length": 60.625, "completions/min_length": 57.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.625, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.5306375622749329, "rewards/meter/std": 0.289710134267807, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5306375622749329, "rewards/total_composite/std": 0.289710134267807, "reward": 0.5306375622749329, "reward_std": 0.289710134267807, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07036992907524109, "sampling/sampling_logp_difference/max": 1.4188556671142578, "sampling/importance_sampling_ratio/min": 0.24199077486991882, "sampling/importance_sampling_ratio/mean": 1.0169787406921387, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6950614899396896, "clip_ratio/low_mean": 0.030917395371943712, "clip_ratio/low_min": 0.030917395371943712, "clip_ratio/high_mean": 0.04732227721251547, "clip_ratio/high_max": 0.04732227721251547, "clip_ratio/region_mean": 0.07823967258445919, "reward_total_mean": 0.5306375622749329, "reward_meter_mean": 0.5306375622749329, "reward_meter_std": 0.289710134267807, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5306375622749329, "reward_total_composite_std": 0.289710134267807} {"timestamp_utc": "2026-04-12T00:20:55Z", "mode": "train", "global_step": 1519, "epoch": 0.06101136683134514, "loss": -0.0293, "grad_norm": 3.1494743824005127, "learning_rate": 5.400000000000001e-06, "num_tokens": 3419414.0, "completions/mean_length": 290.875, "completions/min_length": 274.0, "completions/max_length": 312.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 290.875, "completions/min_terminated_length": 274.0, "completions/max_terminated_length": 312.0, "rewards/meter/mean": 0.9781413078308105, "rewards/meter/std": 0.051752228289842606, "rewards/count_adherence/mean": 0.7045454978942871, "rewards/count_adherence/std": 0.04208274558186531, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.9208333492279053, "rewards/repeat_penalty/std": 0.12346728891134262, "rewards/total_composite/mean": 0.5596551299095154, "rewards/total_composite/std": 0.2521139085292816, "reward": 0.5596551299095154, "reward_std": 0.2521139085292816, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07325157523155212, "sampling/sampling_logp_difference/max": 1.6432204246520996, "sampling/importance_sampling_ratio/min": 0.19335635006427765, "sampling/importance_sampling_ratio/mean": 1.0232716798782349, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7886361554265022, "clip_ratio/low_mean": 0.020800751633942127, "clip_ratio/low_min": 0.020800751633942127, "clip_ratio/high_mean": 0.028688129736110568, "clip_ratio/high_max": 0.028688129736110568, "clip_ratio/region_mean": 0.049488881370052695, "reward_total_mean": 0.5596551299095154, "reward_meter_mean": 0.9781413078308105, "reward_meter_std": 0.051752228289842606, "reward_count_adherence_mean": 0.7045454978942871, "reward_count_adherence_std": 0.04208274558186531, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.9208333492279053, "reward_repeat_penalty_std": 0.12346728891134262, "reward_total_composite_mean": 0.5596551299095154, "reward_total_composite_std": 0.2521139085292816} {"timestamp_utc": "2026-04-12T00:21:00Z", "mode": "train", "global_step": 1520, "epoch": 0.0610515323131301, "loss": 0.0163, "grad_norm": 2.657116413116455, "learning_rate": 5.396969696969697e-06, "num_tokens": 3421603.0, "completions/mean_length": 88.625, "completions/min_length": 86.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 88.625, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.9171510934829712, "rewards/meter/std": 0.1984940767288208, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.8182051777839661, "rewards/total_composite/std": 0.1850285530090332, "reward": 0.8182051777839661, "reward_std": 0.1850285530090332, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01846635341644287, "sampling/sampling_logp_difference/max": 0.759364128112793, "sampling/importance_sampling_ratio/min": 0.46796390414237976, "sampling/importance_sampling_ratio/mean": 1.0104413032531738, "sampling/importance_sampling_ratio/max": 1.6496562957763672, "entropy": 0.1767467763274908, "clip_ratio/low_mean": 0.0028572361916303635, "clip_ratio/low_min": 0.0028572361916303635, "clip_ratio/high_mean": 0.004295865655876696, "clip_ratio/high_max": 0.004295865655876696, "clip_ratio/region_mean": 0.00715310184750706, "reward_total_mean": 0.8182051777839661, "reward_meter_mean": 0.9171510934829712, "reward_meter_std": 0.1984940767288208, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_total_composite_mean": 0.8182051777839661, "reward_total_composite_std": 0.1850285530090332} {"timestamp_utc": "2026-04-12T00:21:04Z", "mode": "train", "global_step": 1521, "epoch": 0.06109169779491505, "loss": 0.0033, "grad_norm": 1.7797563076019287, "learning_rate": 5.3939393939393945e-06, "num_tokens": 3423423.0, "completions/mean_length": 68.5, "completions/min_length": 68.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.5, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9934801459312439, "rewards/meter/std": 0.0036428512539714575, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9934801459312439, "rewards/total_composite/std": 0.0036428512539714575, "reward": 0.9934801459312439, "reward_std": 0.00364283611997962, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01598731242120266, "sampling/sampling_logp_difference/max": 1.9260679483413696, "sampling/importance_sampling_ratio/min": 0.14572004973888397, "sampling/importance_sampling_ratio/mean": 1.0017439126968384, "sampling/importance_sampling_ratio/max": 1.2535090446472168, "entropy": 0.09944538865238428, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.010897728730924428, "clip_ratio/high_max": 0.010897728730924428, "clip_ratio/region_mean": 0.010897728730924428, "reward_total_mean": 0.9934801459312439, "reward_meter_mean": 0.9934801459312439, "reward_meter_std": 0.0036428512539714575, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9934801459312439, "reward_total_composite_std": 0.0036428512539714575} {"timestamp_utc": "2026-04-12T00:21:09Z", "mode": "train", "global_step": 1522, "epoch": 0.061131863276700005, "loss": -0.0022, "grad_norm": 7.3915534019470215, "learning_rate": 5.390909090909091e-06, "num_tokens": 3425367.0, "completions/mean_length": 65.0, "completions/min_length": 63.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.0, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9976125955581665, "rewards/meter/std": 0.0013877107994630933, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9976125955581665, "rewards/total_composite/std": 0.0013877107994630933, "reward": 0.9976125955581665, "reward_std": 0.0013877045130357146, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.040052663534879684, "sampling/sampling_logp_difference/max": 0.6700773239135742, "sampling/importance_sampling_ratio/min": 0.5116690397262573, "sampling/importance_sampling_ratio/mean": 1.0058274269104004, "sampling/importance_sampling_ratio/max": 1.817367672920227, "entropy": 0.37458037585020065, "clip_ratio/low_mean": 0.0058302809484303, "clip_ratio/low_min": 0.0058302809484303, "clip_ratio/high_mean": 0.019181564333848655, "clip_ratio/high_max": 0.019181564333848655, "clip_ratio/region_mean": 0.025011845282278955, "reward_total_mean": 0.9976125955581665, "reward_meter_mean": 0.9976125955581665, "reward_meter_std": 0.0013877107994630933, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9976125955581665, "reward_total_composite_std": 0.0013877107994630933} {"timestamp_utc": "2026-04-12T00:21:13Z", "mode": "train", "global_step": 1523, "epoch": 0.06117202875848496, "loss": 0.0106, "grad_norm": 9.1239013671875, "learning_rate": 5.387878787878789e-06, "num_tokens": 3426883.0, "completions/mean_length": 34.5, "completions/min_length": 32.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.5, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9722611904144287, "rewards/meter/std": 0.0555478073656559, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9722611904144287, "rewards/total_composite/std": 0.0555478073656559, "reward": 0.9722611904144287, "reward_std": 0.05554782226681709, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07421485334634781, "sampling/sampling_logp_difference/max": 1.0678462982177734, "sampling/importance_sampling_ratio/min": 0.3437480628490448, "sampling/importance_sampling_ratio/mean": 1.0279422998428345, "sampling/importance_sampling_ratio/max": 1.5298467874526978, "entropy": 0.7306949496269226, "clip_ratio/low_mean": 0.010714286006987095, "clip_ratio/low_min": 0.010714286006987095, "clip_ratio/high_mean": 0.04335171659477055, "clip_ratio/high_max": 0.04335171659477055, "clip_ratio/region_mean": 0.054066002601757646, "reward_total_mean": 0.9722611904144287, "reward_meter_mean": 0.9722611904144287, "reward_meter_std": 0.0555478073656559, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9722611904144287, "reward_total_composite_std": 0.0555478073656559} {"timestamp_utc": "2026-04-12T00:21:19Z", "mode": "train", "global_step": 1524, "epoch": 0.06121219424026991, "loss": 0.0095, "grad_norm": 5.130848407745361, "learning_rate": 5.384848484848485e-06, "num_tokens": 3429089.0, "completions/mean_length": 108.75, "completions/min_length": 99.0, "completions/max_length": 113.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.75, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 113.0, "rewards/meter/mean": 0.9915084838867188, "rewards/meter/std": 0.007808292284607887, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9915084838867188, "rewards/total_composite/std": 0.007808292284607887, "reward": 0.9915084838867188, "reward_std": 0.007808296009898186, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08183825016021729, "sampling/sampling_logp_difference/max": 1.6284008026123047, "sampling/importance_sampling_ratio/min": 0.19624315202236176, "sampling/importance_sampling_ratio/mean": 1.0207182168960571, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.894000805914402, "clip_ratio/low_mean": 0.01814199541695416, "clip_ratio/low_min": 0.01814199541695416, "clip_ratio/high_mean": 0.03156230039894581, "clip_ratio/high_max": 0.03156230039894581, "clip_ratio/region_mean": 0.04970429581589997, "reward_total_mean": 0.9915084838867188, "reward_meter_mean": 0.9915084838867188, "reward_meter_std": 0.007808292284607887, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9915084838867188, "reward_total_composite_std": 0.007808292284607887} {"timestamp_utc": "2026-04-12T00:21:24Z", "mode": "train", "global_step": 1525, "epoch": 0.06125235972205487, "loss": 0.0026, "grad_norm": 2.808112382888794, "learning_rate": 5.381818181818183e-06, "num_tokens": 3431040.0, "completions/mean_length": 86.875, "completions/min_length": 86.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.875, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.99290531873703, "rewards/meter/std": 0.0034807121846824884, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.9431794881820679, "rewards/total_composite/std": 0.09114136546850204, "reward": 0.9431794881820679, "reward_std": 0.09114135801792145, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017489619553089142, "sampling/sampling_logp_difference/max": 0.7286503314971924, "sampling/importance_sampling_ratio/min": 0.48398464918136597, "sampling/importance_sampling_ratio/mean": 1.0029078722000122, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09439879935234785, "clip_ratio/low_mean": 0.0028735632076859474, "clip_ratio/low_min": 0.0028735632076859474, "clip_ratio/high_mean": 0.012915466912090778, "clip_ratio/high_max": 0.012915466912090778, "clip_ratio/region_mean": 0.015789030119776726, "reward_total_mean": 0.9431794881820679, "reward_meter_mean": 0.99290531873703, "reward_meter_std": 0.0034807121846824884, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.9431794881820679, "reward_total_composite_std": 0.09114136546850204} {"timestamp_utc": "2026-04-12T00:21:29Z", "mode": "train", "global_step": 1526, "epoch": 0.06129252520383982, "loss": -0.0074, "grad_norm": 2.34173321723938, "learning_rate": 5.378787878787879e-06, "num_tokens": 3433169.0, "completions/mean_length": 95.125, "completions/min_length": 92.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.125, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9933592081069946, "rewards/meter/std": 0.00307702855207026, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8250000476837158, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8195037841796875, "rewards/total_composite/std": 0.07000674307346344, "reward": 0.8195037841796875, "reward_std": 0.07000675052404404, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023438630625605583, "sampling/sampling_logp_difference/max": 1.2709712982177734, "sampling/importance_sampling_ratio/min": 0.2805590033531189, "sampling/importance_sampling_ratio/mean": 1.006750464439392, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.13843786902725697, "clip_ratio/low_mean": 0.017182219657115638, "clip_ratio/low_min": 0.017182219657115638, "clip_ratio/high_mean": 0.0025510203558951616, "clip_ratio/high_max": 0.0025510203558951616, "clip_ratio/region_mean": 0.0197332400130108, "reward_total_mean": 0.8195037841796875, "reward_meter_mean": 0.9933592081069946, "reward_meter_std": 0.00307702855207026, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8250000476837158, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.8195037841796875, "reward_total_composite_std": 0.07000674307346344} {"timestamp_utc": "2026-04-12T00:21:34Z", "mode": "train", "global_step": 1527, "epoch": 0.061332690685624774, "loss": 0.0164, "grad_norm": 3.2660653591156006, "learning_rate": 5.375757575757576e-06, "num_tokens": 3435703.0, "completions/mean_length": 142.75, "completions/min_length": 134.0, "completions/max_length": 149.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 142.75, "completions/min_terminated_length": 134.0, "completions/max_terminated_length": 149.0, "rewards/meter/mean": 0.994535505771637, "rewards/meter/std": 0.0031923020724207163, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.10628911107778549, "rewards/total_composite/mean": 0.9411816596984863, "rewards/total_composite/std": 0.10509292781352997, "reward": 0.9411816596984863, "reward_std": 0.10509292781352997, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07457228004932404, "sampling/sampling_logp_difference/max": 1.6383848190307617, "sampling/importance_sampling_ratio/min": 0.19429361820220947, "sampling/importance_sampling_ratio/mean": 1.02262282371521, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7129531875252724, "clip_ratio/low_mean": 0.006133304443210363, "clip_ratio/low_min": 0.006133304443210363, "clip_ratio/high_mean": 0.044742210768163204, "clip_ratio/high_max": 0.044742210768163204, "clip_ratio/region_mean": 0.05087551521137357, "reward_total_mean": 0.9411816596984863, "reward_meter_mean": 0.994535505771637, "reward_meter_std": 0.0031923020724207163, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.10628911107778549, "reward_total_composite_mean": 0.9411816596984863, "reward_total_composite_std": 0.10509292781352997} {"timestamp_utc": "2026-04-12T00:21:39Z", "mode": "train", "global_step": 1528, "epoch": 0.06137285616740973, "loss": -0.0161, "grad_norm": 5.586496353149414, "learning_rate": 5.372727272727273e-06, "num_tokens": 3437794.0, "completions/mean_length": 93.375, "completions/min_length": 89.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.375, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.8450676202774048, "rewards/meter/std": 0.220209538936615, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8450676202774048, "rewards/total_composite/std": 0.220209538936615, "reward": 0.8450676202774048, "reward_std": 0.2202095091342926, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05679401755332947, "sampling/sampling_logp_difference/max": 0.9821338653564453, "sampling/importance_sampling_ratio/min": 0.37451109290122986, "sampling/importance_sampling_ratio/mean": 1.0161728858947754, "sampling/importance_sampling_ratio/max": 1.8447113037109375, "entropy": 0.5462022610008717, "clip_ratio/low_mean": 0.009738856926560402, "clip_ratio/low_min": 0.009738856926560402, "clip_ratio/high_mean": 0.02779570873826742, "clip_ratio/high_max": 0.02779570873826742, "clip_ratio/region_mean": 0.037534565664827824, "reward_total_mean": 0.8450676202774048, "reward_meter_mean": 0.8450676202774048, "reward_meter_std": 0.220209538936615, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8450676202774048, "reward_total_composite_std": 0.220209538936615} {"timestamp_utc": "2026-04-12T00:21:44Z", "mode": "train", "global_step": 1529, "epoch": 0.06141302164919468, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.36969696969697e-06, "num_tokens": 3439074.0, "completions/mean_length": 33.0, "completions/min_length": 33.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9988001585006714, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9988001585006714, "rewards/total_composite/std": 0.0, "reward": 0.9988001585006714, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0021949801594018936, "sampling/sampling_logp_difference/max": 0.04461796581745148, "sampling/importance_sampling_ratio/min": 0.9977096915245056, "sampling/importance_sampling_ratio/mean": 1.0021947622299194, "sampling/importance_sampling_ratio/max": 1.045628309249878, "entropy": 0.017181317321956158, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9988001585006714, "reward_meter_mean": 0.9988001585006714, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9988001585006714, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T00:21:48Z", "mode": "train", "global_step": 1530, "epoch": 0.061453187130979636, "loss": 0.0132, "grad_norm": 3.0035605430603027, "learning_rate": 5.366666666666666e-06, "num_tokens": 3440826.0, "completions/mean_length": 66.0, "completions/min_length": 64.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9936773180961609, "rewards/meter/std": 0.0019825000781565905, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9936773180961609, "rewards/total_composite/std": 0.0019825000781565905, "reward": 0.9936773180961609, "reward_std": 0.001982480986043811, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02317051589488983, "sampling/sampling_logp_difference/max": 0.8188035488128662, "sampling/importance_sampling_ratio/min": 0.4409589171409607, "sampling/importance_sampling_ratio/mean": 1.0015543699264526, "sampling/importance_sampling_ratio/max": 1.562909722328186, "entropy": 0.1264043264091015, "clip_ratio/low_mean": 0.003791360300965607, "clip_ratio/low_min": 0.003791360300965607, "clip_ratio/high_mean": 0.015155904809944332, "clip_ratio/high_max": 0.015155904809944332, "clip_ratio/region_mean": 0.01894726511090994, "reward_total_mean": 0.9936773180961609, "reward_meter_mean": 0.9936773180961609, "reward_meter_std": 0.0019825000781565905, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9936773180961609, "reward_total_composite_std": 0.0019825000781565905} {"timestamp_utc": "2026-04-12T00:21:54Z", "mode": "train", "global_step": 1531, "epoch": 0.06149335261276459, "loss": -0.0098, "grad_norm": 5.306570053100586, "learning_rate": 5.3636363636363645e-06, "num_tokens": 3443097.0, "completions/mean_length": 106.875, "completions/min_length": 104.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.875, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.980573296546936, "rewards/meter/std": 0.03288979455828667, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8351579904556274, "rewards/total_composite/std": 0.3450130224227905, "reward": 0.8351579904556274, "reward_std": 0.3450130224227905, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08104944229125977, "sampling/sampling_logp_difference/max": 2.037365436553955, "sampling/importance_sampling_ratio/min": 0.13037173449993134, "sampling/importance_sampling_ratio/mean": 1.011287808418274, "sampling/importance_sampling_ratio/max": 1.9533883333206177, "entropy": 0.799922339618206, "clip_ratio/low_mean": 0.007166183087974787, "clip_ratio/low_min": 0.007166183087974787, "clip_ratio/high_mean": 0.045307138469070196, "clip_ratio/high_max": 0.045307138469070196, "clip_ratio/region_mean": 0.05247332155704498, "reward_total_mean": 0.8351579904556274, "reward_meter_mean": 0.980573296546936, "reward_meter_std": 0.03288979455828667, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.8351579904556274, "reward_total_composite_std": 0.3450130224227905} {"timestamp_utc": "2026-04-12T00:21:59Z", "mode": "train", "global_step": 1532, "epoch": 0.061533518094549544, "loss": -0.0044, "grad_norm": 3.583285331726074, "learning_rate": 5.360606060606061e-06, "num_tokens": 3445821.0, "completions/mean_length": 160.5, "completions/min_length": 155.0, "completions/max_length": 165.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 160.5, "completions/min_terminated_length": 155.0, "completions/max_terminated_length": 165.0, "rewards/meter/mean": 0.8937799334526062, "rewards/meter/std": 0.1616331785917282, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8333333134651184, "rewards/repeat_penalty/std": 0.11878276616334915, "rewards/total_composite/mean": 0.7505040168762207, "rewards/total_composite/std": 0.18041202425956726, "reward": 0.7505040168762207, "reward_std": 0.18041202425956726, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06113876402378082, "sampling/sampling_logp_difference/max": 1.3376035690307617, "sampling/importance_sampling_ratio/min": 0.2624739110469818, "sampling/importance_sampling_ratio/mean": 1.012855052947998, "sampling/importance_sampling_ratio/max": 1.947861909866333, "entropy": 0.5020967051386833, "clip_ratio/low_mean": 0.01080156397074461, "clip_ratio/low_min": 0.01080156397074461, "clip_ratio/high_mean": 0.035965551156550646, "clip_ratio/high_max": 0.035965551156550646, "clip_ratio/region_mean": 0.046767115127295256, "reward_total_mean": 0.7505040168762207, "reward_meter_mean": 0.8937799334526062, "reward_meter_std": 0.1616331785917282, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8333333134651184, "reward_repeat_penalty_std": 0.11878276616334915, "reward_total_composite_mean": 0.7505040168762207, "reward_total_composite_std": 0.18041202425956726} {"timestamp_utc": "2026-04-12T00:22:04Z", "mode": "train", "global_step": 1533, "epoch": 0.0615736835763345, "loss": 0.0012, "grad_norm": 4.299396514892578, "learning_rate": 5.357575757575758e-06, "num_tokens": 3447376.0, "completions/mean_length": 34.375, "completions/min_length": 33.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.375, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9954771995544434, "rewards/meter/std": 0.002512240083888173, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9954771995544434, "rewards/total_composite/std": 0.002512240083888173, "reward": 0.9954771995544434, "reward_std": 0.00251222332008183, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01558676641434431, "sampling/sampling_logp_difference/max": 0.46267223358154297, "sampling/importance_sampling_ratio/min": 0.7345414757728577, "sampling/importance_sampling_ratio/mean": 1.009590983390808, "sampling/importance_sampling_ratio/max": 1.5883126258850098, "entropy": 0.13886073511093855, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/high_mean": 0.010924369795247912, "clip_ratio/high_max": 0.010924369795247912, "clip_ratio/region_mean": 0.014712248696014285, "reward_total_mean": 0.9954771995544434, "reward_meter_mean": 0.9954771995544434, "reward_meter_std": 0.002512240083888173, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9954771995544434, "reward_total_composite_std": 0.002512240083888173} {"timestamp_utc": "2026-04-12T00:22:13Z", "mode": "train", "global_step": 1534, "epoch": 0.06161384905811945, "loss": -0.1403, "grad_norm": 1.0674489736557007, "learning_rate": 5.3545454545454546e-06, "num_tokens": 3449061.0, "completions/mean_length": 106.625, "completions/min_length": 45.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 48.71428680419922, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.8021861910820007, "rewards/meter/std": 0.325016587972641, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.80141282081604, "rewards/total_composite/std": 0.327181339263916, "reward": 0.80141282081604, "reward_std": 0.32718130946159363, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.038483839482069016, "sampling/sampling_logp_difference/max": 2.473947525024414, "sampling/importance_sampling_ratio/min": 0.08425161987543106, "sampling/importance_sampling_ratio/mean": 1.0069955587387085, "sampling/importance_sampling_ratio/max": 1.8218151330947876, "entropy": 0.16696601919829845, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.023747874423861504, "clip_ratio/high_max": 0.023747874423861504, "clip_ratio/region_mean": 0.023747874423861504, "reward_total_mean": 0.80141282081604, "reward_meter_mean": 0.8021861910820007, "reward_meter_std": 0.325016587972641, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.80141282081604, "reward_total_composite_std": 0.327181339263916} {"timestamp_utc": "2026-04-12T00:22:19Z", "mode": "train", "global_step": 1535, "epoch": 0.061654014539904406, "loss": -0.0071, "grad_norm": 4.45563268661499, "learning_rate": 5.351515151515152e-06, "num_tokens": 3451653.0, "completions/mean_length": 153.0, "completions/min_length": 130.0, "completions/max_length": 174.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 153.0, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 174.0, "rewards/meter/mean": 0.7749242782592773, "rewards/meter/std": 0.3511018455028534, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.1035098284482956, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.07856741547584534, "rewards/total_composite/mean": 0.6905776858329773, "rewards/total_composite/std": 0.3377683758735657, "reward": 0.6905776858329773, "reward_std": 0.3377683758735657, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05948156490921974, "sampling/sampling_logp_difference/max": 1.3245697021484375, "sampling/importance_sampling_ratio/min": 0.2659173607826233, "sampling/importance_sampling_ratio/mean": 1.0109103918075562, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5641567148268223, "clip_ratio/low_mean": 0.018939394503831863, "clip_ratio/low_min": 0.018939394503831863, "clip_ratio/high_mean": 0.0292887045070529, "clip_ratio/high_max": 0.0292887045070529, "clip_ratio/region_mean": 0.04822809901088476, "reward_total_mean": 0.6905776858329773, "reward_meter_mean": 0.7749242782592773, "reward_meter_std": 0.3511018455028534, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.1035098284482956, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.07856741547584534, "reward_total_composite_mean": 0.6905776858329773, "reward_total_composite_std": 0.3377683758735657} {"timestamp_utc": "2026-04-12T00:22:24Z", "mode": "train", "global_step": 1536, "epoch": 0.06169418002168936, "loss": -0.0035, "grad_norm": 4.13173770904541, "learning_rate": 5.348484848484848e-06, "num_tokens": 3453309.0, "completions/mean_length": 68.0, "completions/min_length": 66.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.8998743295669556, "rewards/meter/std": 0.1591634452342987, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8998743295669556, "rewards/total_composite/std": 0.1591634452342987, "reward": 0.8998743295669556, "reward_std": 0.1591634303331375, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05023188889026642, "sampling/sampling_logp_difference/max": 1.0234405994415283, "sampling/importance_sampling_ratio/min": 0.35935643315315247, "sampling/importance_sampling_ratio/mean": 1.0069003105163574, "sampling/importance_sampling_ratio/max": 1.8585302829742432, "entropy": 0.4786563478410244, "clip_ratio/low_mean": 0.016904115676879883, "clip_ratio/low_min": 0.016904115676879883, "clip_ratio/high_mean": 0.03292302007321268, "clip_ratio/high_max": 0.03292302007321268, "clip_ratio/region_mean": 0.049827135750092566, "reward_total_mean": 0.8998743295669556, "reward_meter_mean": 0.8998743295669556, "reward_meter_std": 0.1591634452342987, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8998743295669556, "reward_total_composite_std": 0.1591634452342987} {"timestamp_utc": "2026-04-12T00:22:28Z", "mode": "train", "global_step": 1537, "epoch": 0.061734345503474314, "loss": 0.0252, "grad_norm": 8.413439750671387, "learning_rate": 5.3454545454545455e-06, "num_tokens": 3454809.0, "completions/mean_length": 30.5, "completions/min_length": 29.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.5, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9885121583938599, "rewards/meter/std": 0.006495221517980099, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9885121583938599, "rewards/total_composite/std": 0.006495221517980099, "reward": 0.9885121583938599, "reward_std": 0.006495220120996237, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03234010562300682, "sampling/sampling_logp_difference/max": 0.9072532653808594, "sampling/importance_sampling_ratio/min": 0.40363138914108276, "sampling/importance_sampling_ratio/mean": 1.0021849870681763, "sampling/importance_sampling_ratio/max": 1.6943132877349854, "entropy": 0.229267543181777, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/high_mean": 0.01667593652382493, "clip_ratio/high_max": 0.01667593652382493, "clip_ratio/region_mean": 0.02058218652382493, "reward_total_mean": 0.9885121583938599, "reward_meter_mean": 0.9885121583938599, "reward_meter_std": 0.006495221517980099, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9885121583938599, "reward_total_composite_std": 0.006495221517980099} {"timestamp_utc": "2026-04-12T00:22:33Z", "mode": "train", "global_step": 1538, "epoch": 0.06177451098525927, "loss": -0.0059, "grad_norm": 3.6924545764923096, "learning_rate": 5.342424242424244e-06, "num_tokens": 3456577.0, "completions/mean_length": 67.0, "completions/min_length": 65.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9983944892883301, "rewards/meter/std": 0.0004069809801876545, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9983944892883301, "rewards/total_composite/std": 0.0004069809801876545, "reward": 0.9983944892883301, "reward_std": 0.00040697952499613166, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0381365530192852, "sampling/sampling_logp_difference/max": 1.0573501586914062, "sampling/importance_sampling_ratio/min": 0.3473750650882721, "sampling/importance_sampling_ratio/mean": 1.0129139423370361, "sampling/importance_sampling_ratio/max": 1.5474638938903809, "entropy": 0.3516472205519676, "clip_ratio/low_mean": 0.007634902372956276, "clip_ratio/low_min": 0.007634902372956276, "clip_ratio/high_mean": 0.018546971026808023, "clip_ratio/high_max": 0.018546971026808023, "clip_ratio/region_mean": 0.0261818733997643, "reward_total_mean": 0.9983944892883301, "reward_meter_mean": 0.9983944892883301, "reward_meter_std": 0.0004069809801876545, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9983944892883301, "reward_total_composite_std": 0.0004069809801876545} {"timestamp_utc": "2026-04-12T00:22:38Z", "mode": "train", "global_step": 1539, "epoch": 0.06181467646704422, "loss": -0.0048, "grad_norm": 2.03717303276062, "learning_rate": 5.33939393939394e-06, "num_tokens": 3458618.0, "completions/mean_length": 87.125, "completions/min_length": 86.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.125, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9949196577072144, "rewards/meter/std": 0.00045097857946529984, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9949196577072144, "rewards/total_composite/std": 0.00045097857946529984, "reward": 0.9949196577072144, "reward_std": 0.00045097683323547244, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0125720901414752, "sampling/sampling_logp_difference/max": 1.0620005130767822, "sampling/importance_sampling_ratio/min": 0.3457634150981903, "sampling/importance_sampling_ratio/mean": 1.0028852224349976, "sampling/importance_sampling_ratio/max": 1.4961625337600708, "entropy": 0.08502828422933817, "clip_ratio/low_mean": 0.0014367816038429737, "clip_ratio/low_min": 0.0014367816038429737, "clip_ratio/high_mean": 0.009995099389925599, "clip_ratio/high_max": 0.009995099389925599, "clip_ratio/region_mean": 0.011431880993768573, "reward_total_mean": 0.9949196577072144, "reward_meter_mean": 0.9949196577072144, "reward_meter_std": 0.00045097857946529984, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9949196577072144, "reward_total_composite_std": 0.00045097857946529984} {"timestamp_utc": "2026-04-12T00:22:42Z", "mode": "train", "global_step": 1540, "epoch": 0.061854841948829176, "loss": -0.0382, "grad_norm": 12.248023986816406, "learning_rate": 5.336363636363637e-06, "num_tokens": 3460405.0, "completions/mean_length": 48.375, "completions/min_length": 41.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.375, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.8395511507987976, "rewards/meter/std": 0.27874672412872314, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8395511507987976, "rewards/total_composite/std": 0.27874672412872314, "reward": 0.8395511507987976, "reward_std": 0.27874669432640076, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04395786300301552, "sampling/sampling_logp_difference/max": 1.4850921630859375, "sampling/importance_sampling_ratio/min": 0.35930031538009644, "sampling/importance_sampling_ratio/mean": 1.0040042400360107, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2700903480872512, "clip_ratio/low_mean": 0.009146341122686863, "clip_ratio/low_min": 0.009146341122686863, "clip_ratio/high_mean": 0.023029743460938334, "clip_ratio/high_max": 0.023029743460938334, "clip_ratio/region_mean": 0.0321760845836252, "reward_total_mean": 0.8395511507987976, "reward_meter_mean": 0.8395511507987976, "reward_meter_std": 0.27874672412872314, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8395511507987976, "reward_total_composite_std": 0.27874672412872314} {"timestamp_utc": "2026-04-12T00:22:47Z", "mode": "train", "global_step": 1541, "epoch": 0.06189500743061413, "loss": 0.0077, "grad_norm": 7.992240905761719, "learning_rate": 5.333333333333334e-06, "num_tokens": 3462213.0, "completions/mean_length": 60.0, "completions/min_length": 58.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.0, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9441971778869629, "rewards/meter/std": 0.13217569887638092, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9441971778869629, "rewards/total_composite/std": 0.13217569887638092, "reward": 0.9441971778869629, "reward_std": 0.13217571377754211, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06318628787994385, "sampling/sampling_logp_difference/max": 1.5191354751586914, "sampling/importance_sampling_ratio/min": 0.21890106797218323, "sampling/importance_sampling_ratio/mean": 1.0145553350448608, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47805059514939785, "clip_ratio/low_mean": 0.008196720853447914, "clip_ratio/low_min": 0.008196720853447914, "clip_ratio/high_mean": 0.0310292961075902, "clip_ratio/high_max": 0.0310292961075902, "clip_ratio/region_mean": 0.03922601696103811, "reward_total_mean": 0.9441971778869629, "reward_meter_mean": 0.9441971778869629, "reward_meter_std": 0.13217569887638092, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9441971778869629, "reward_total_composite_std": 0.13217569887638092} {"timestamp_utc": "2026-04-12T00:22:52Z", "mode": "train", "global_step": 1542, "epoch": 0.061935172912399084, "loss": 0.0082, "grad_norm": 3.4401650428771973, "learning_rate": 5.330303030303031e-06, "num_tokens": 3464650.0, "completions/mean_length": 127.625, "completions/min_length": 124.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.625, "completions/min_terminated_length": 124.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.9891485571861267, "rewards/meter/std": 0.004043546039611101, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9464285373687744, "rewards/repeat_penalty/std": 0.10628911107778549, "rewards/total_composite/mean": 0.9360376596450806, "rewards/total_composite/std": 0.10430651903152466, "reward": 0.9360376596450806, "reward_std": 0.10430650413036346, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07292276620864868, "sampling/sampling_logp_difference/max": 1.6305036544799805, "sampling/importance_sampling_ratio/min": 0.19583091139793396, "sampling/importance_sampling_ratio/mean": 1.0177416801452637, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6336969211697578, "clip_ratio/low_mean": 0.005937984678894281, "clip_ratio/low_min": 0.005937984678894281, "clip_ratio/high_mean": 0.04202882433310151, "clip_ratio/high_max": 0.04202882433310151, "clip_ratio/region_mean": 0.04796680901199579, "reward_total_mean": 0.9360376596450806, "reward_meter_mean": 0.9891485571861267, "reward_meter_std": 0.004043546039611101, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9464285373687744, "reward_repeat_penalty_std": 0.10628911107778549, "reward_total_composite_mean": 0.9360376596450806, "reward_total_composite_std": 0.10430651903152466} {"timestamp_utc": "2026-04-12T00:22:58Z", "mode": "train", "global_step": 1543, "epoch": 0.06197533839418404, "loss": -0.0036, "grad_norm": 5.007582664489746, "learning_rate": 5.327272727272727e-06, "num_tokens": 3466723.0, "completions/mean_length": 96.125, "completions/min_length": 93.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.125, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.7870810627937317, "rewards/meter/std": 0.3985043168067932, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7870810627937317, "rewards/total_composite/std": 0.3985043168067932, "reward": 0.7870810627937317, "reward_std": 0.3985043168067932, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026287445798516273, "sampling/sampling_logp_difference/max": 1.5934228897094727, "sampling/importance_sampling_ratio/min": 0.20322878658771515, "sampling/importance_sampling_ratio/mean": 0.9980475306510925, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08730749879032373, "clip_ratio/low_mean": 0.00649872503709048, "clip_ratio/low_min": 0.00649872503709048, "clip_ratio/high_mean": 0.009101442527025938, "clip_ratio/high_max": 0.009101442527025938, "clip_ratio/region_mean": 0.015600167564116418, "reward_total_mean": 0.7870810627937317, "reward_meter_mean": 0.7870810627937317, "reward_meter_std": 0.3985043168067932, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7870810627937317, "reward_total_composite_std": 0.3985043168067932} {"timestamp_utc": "2026-04-12T00:23:02Z", "mode": "train", "global_step": 1544, "epoch": 0.06201550387596899, "loss": -0.0189, "grad_norm": 4.138071060180664, "learning_rate": 5.324242424242425e-06, "num_tokens": 3468547.0, "completions/mean_length": 65.0, "completions/min_length": 62.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9653145670890808, "rewards/meter/std": 0.03330826759338379, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9653145670890808, "rewards/total_composite/std": 0.03330826759338379, "reward": 0.9653145670890808, "reward_std": 0.03330826386809349, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025457728654146194, "sampling/sampling_logp_difference/max": 0.9563794136047363, "sampling/importance_sampling_ratio/min": 0.4773949682712555, "sampling/importance_sampling_ratio/mean": 1.0086570978164673, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.14403609465807676, "clip_ratio/low_mean": 0.007875503972172737, "clip_ratio/low_min": 0.007875503972172737, "clip_ratio/high_mean": 0.01856476883403957, "clip_ratio/high_max": 0.01856476883403957, "clip_ratio/region_mean": 0.026440272806212306, "reward_total_mean": 0.9653145670890808, "reward_meter_mean": 0.9653145670890808, "reward_meter_std": 0.03330826759338379, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9653145670890808, "reward_total_composite_std": 0.03330826759338379} {"timestamp_utc": "2026-04-12T00:23:07Z", "mode": "train", "global_step": 1545, "epoch": 0.062055669357753945, "loss": -0.0078, "grad_norm": 5.288153648376465, "learning_rate": 5.321212121212122e-06, "num_tokens": 3470357.0, "completions/mean_length": 67.25, "completions/min_length": 62.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.25, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9962025284767151, "rewards/meter/std": 0.00486297020688653, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9962025284767151, "rewards/total_composite/std": 0.00486297020688653, "reward": 0.9962025284767151, "reward_std": 0.004862963687628508, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.038305725902318954, "sampling/sampling_logp_difference/max": 1.3244657516479492, "sampling/importance_sampling_ratio/min": 0.2659450173377991, "sampling/importance_sampling_ratio/mean": 1.0034021139144897, "sampling/importance_sampling_ratio/max": 1.475104570388794, "entropy": 0.33554743882268667, "clip_ratio/low_mean": 0.01173020526766777, "clip_ratio/low_min": 0.01173020526766777, "clip_ratio/high_mean": 0.02582784171681851, "clip_ratio/high_max": 0.02582784171681851, "clip_ratio/region_mean": 0.03755804698448628, "reward_total_mean": 0.9962025284767151, "reward_meter_mean": 0.9962025284767151, "reward_meter_std": 0.00486297020688653, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9962025284767151, "reward_total_composite_std": 0.00486297020688653} {"timestamp_utc": "2026-04-12T00:23:13Z", "mode": "train", "global_step": 1546, "epoch": 0.0620958348395389, "loss": -0.0044, "grad_norm": 2.919037342071533, "learning_rate": 5.318181818181819e-06, "num_tokens": 3472779.0, "completions/mean_length": 127.75, "completions/min_length": 122.0, "completions/max_length": 134.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.75, "completions/min_terminated_length": 122.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.9427036046981812, "rewards/meter/std": 0.14980448782444, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.6911537647247314, "rewards/total_composite/std": 0.12467769533395767, "reward": 0.6911537647247314, "reward_std": 0.12467768788337708, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022355277091264725, "sampling/sampling_logp_difference/max": 1.9683294296264648, "sampling/importance_sampling_ratio/min": 0.13969002664089203, "sampling/importance_sampling_ratio/mean": 1.0017569065093994, "sampling/importance_sampling_ratio/max": 1.6036888360977173, "entropy": 0.14278420340269804, "clip_ratio/low_mean": 0.005859375, "clip_ratio/low_min": 0.005859375, "clip_ratio/high_mean": 0.013565883273258805, "clip_ratio/high_max": 0.013565883273258805, "clip_ratio/region_mean": 0.019425258273258805, "reward_total_mean": 0.6911537647247314, "reward_meter_mean": 0.9427036046981812, "reward_meter_std": 0.14980448782444, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.6911537647247314, "reward_total_composite_std": 0.12467769533395767} {"timestamp_utc": "2026-04-12T00:23:17Z", "mode": "train", "global_step": 1547, "epoch": 0.06213600032132385, "loss": -0.0066, "grad_norm": 6.456394195556641, "learning_rate": 5.3151515151515155e-06, "num_tokens": 3474543.0, "completions/mean_length": 60.5, "completions/min_length": 57.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.5, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.796026885509491, "rewards/meter/std": 0.2907305955886841, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6942555904388428, "rewards/total_composite/std": 0.4039342701435089, "reward": 0.6942555904388428, "reward_std": 0.4039342701435089, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07712767273187637, "sampling/sampling_logp_difference/max": 1.3398246765136719, "sampling/importance_sampling_ratio/min": 0.26189157366752625, "sampling/importance_sampling_ratio/mean": 1.0131072998046875, "sampling/importance_sampling_ratio/max": 1.528890609741211, "entropy": 0.7294313032180071, "clip_ratio/low_mean": 0.015021929983049631, "clip_ratio/low_min": 0.015021929983049631, "clip_ratio/high_mean": 0.04056679271161556, "clip_ratio/high_max": 0.04056679271161556, "clip_ratio/region_mean": 0.055588722694665194, "reward_total_mean": 0.6942555904388428, "reward_meter_mean": 0.796026885509491, "reward_meter_std": 0.2907305955886841, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6942555904388428, "reward_total_composite_std": 0.4039342701435089} {"timestamp_utc": "2026-04-12T00:23:22Z", "mode": "train", "global_step": 1548, "epoch": 0.06217616580310881, "loss": 0.0213, "grad_norm": 11.082610130310059, "learning_rate": 5.312121212121213e-06, "num_tokens": 3476009.0, "completions/mean_length": 34.25, "completions/min_length": 32.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.25, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.8179029822349548, "rewards/meter/std": 0.24513599276542664, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8179029822349548, "rewards/total_composite/std": 0.24513599276542664, "reward": 0.8179029822349548, "reward_std": 0.24513597786426544, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07286211103200912, "sampling/sampling_logp_difference/max": 1.024322509765625, "sampling/importance_sampling_ratio/min": 0.35903966426849365, "sampling/importance_sampling_ratio/mean": 1.0309749841690063, "sampling/importance_sampling_ratio/max": 1.7788945436477661, "entropy": 0.7250464260578156, "clip_ratio/low_mean": 0.007575757801532745, "clip_ratio/low_min": 0.007575757801532745, "clip_ratio/high_mean": 0.03996537090279162, "clip_ratio/high_max": 0.03996537090279162, "clip_ratio/region_mean": 0.047541128704324365, "reward_total_mean": 0.8179029822349548, "reward_meter_mean": 0.8179029822349548, "reward_meter_std": 0.24513599276542664, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8179029822349548, "reward_total_composite_std": 0.24513599276542664} {"timestamp_utc": "2026-04-12T00:23:26Z", "mode": "train", "global_step": 1549, "epoch": 0.06221633128489376, "loss": -0.0072, "grad_norm": 8.7092924118042, "learning_rate": 5.309090909090909e-06, "num_tokens": 3477433.0, "completions/mean_length": 30.0, "completions/min_length": 30.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.0, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.9894814491271973, "rewards/meter/std": 0.00789269246160984, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9894814491271973, "rewards/total_composite/std": 0.00789269246160984, "reward": 0.9894814491271973, "reward_std": 0.007892684079706669, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014411688782274723, "sampling/sampling_logp_difference/max": 1.21435546875, "sampling/importance_sampling_ratio/min": 0.29690131545066833, "sampling/importance_sampling_ratio/mean": 0.9964293241500854, "sampling/importance_sampling_ratio/max": 1.0616073608398438, "entropy": 0.06372858118265867, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9894814491271973, "reward_meter_mean": 0.9894814491271973, "reward_meter_std": 0.00789269246160984, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9894814491271973, "reward_total_composite_std": 0.00789269246160984} {"timestamp_utc": "2026-04-12T00:23:31Z", "mode": "train", "global_step": 1550, "epoch": 0.062256496766678715, "loss": 0.0078, "grad_norm": 4.305907249450684, "learning_rate": 5.306060606060606e-06, "num_tokens": 3479115.0, "completions/mean_length": 58.25, "completions/min_length": 56.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.25, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9904994964599609, "rewards/meter/std": 0.009607957676053047, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7916666865348816, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.7832661867141724, "rewards/total_composite/std": 0.1659567952156067, "reward": 0.7832661867141724, "reward_std": 0.1659567803144455, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01842333748936653, "sampling/sampling_logp_difference/max": 0.9088270664215088, "sampling/importance_sampling_ratio/min": 0.40299662947654724, "sampling/importance_sampling_ratio/mean": 1.000774621963501, "sampling/importance_sampling_ratio/max": 1.4585957527160645, "entropy": 0.09011476766318083, "clip_ratio/low_mean": 0.004273816477507353, "clip_ratio/low_min": 0.004273816477507353, "clip_ratio/high_mean": 0.010820218129083514, "clip_ratio/high_max": 0.010820218129083514, "clip_ratio/region_mean": 0.015094034606590867, "reward_total_mean": 0.7832661867141724, "reward_meter_mean": 0.9904994964599609, "reward_meter_std": 0.009607957676053047, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7916666865348816, "reward_repeat_penalty_std": 0.17251639068126678, "reward_total_composite_mean": 0.7832661867141724, "reward_total_composite_std": 0.1659567952156067} {"timestamp_utc": "2026-04-12T00:24:30Z", "mode": "eval", "global_step": 1550, "epoch": 0.062256496766678715, "eval_loss": NaN, "eval_runtime": 59.4299, "eval_samples_per_second": 1.75, "eval_steps_per_second": 0.219, "eval_num_tokens": 3479115.0, "eval_completions/mean_length": 174.25, "eval_completions/min_length": 62.30769230769231, "eval_completions/max_length": 299.7692307692308, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 174.25, "eval_completions/min_terminated_length": 62.30769230769231, "eval_completions/max_terminated_length": 299.7692307692308, "eval_rewards/meter/mean": 0.7260088416246268, "eval_rewards/meter/std": 0.3893573456085645, "eval_rewards/count_adherence/mean": 0.8626743646768423, "eval_rewards/count_adherence/std": 0.15295347571372986, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/repeat_penalty/mean": 0.8659125016285822, "eval_rewards/repeat_penalty/std": 0.14054490625858307, "eval_rewards/total_composite/mean": 0.5367342164883246, "eval_rewards/total_composite/std": 0.3409125472490604, "eval_reward": 0.5367342164883246, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.032458307078251473, "eval_sampling/sampling_logp_difference/max": 1.1005325317382812, "eval_sampling/importance_sampling_ratio/min": 0.34075071719976574, "eval_sampling/importance_sampling_ratio/mean": 1.0097044522945697, "eval_sampling/importance_sampling_ratio/max": 1.5303159401966975, "eval_entropy": 0.3527725820357983, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5367342164883246, "eval_reward_meter_mean": 0.7260088416246268, "eval_reward_meter_std": 0.3893573456085645, "eval_reward_count_adherence_mean": 0.8626743646768423, "eval_reward_count_adherence_std": 0.15295347571372986, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_repeat_penalty_mean": 0.8659125016285822, "eval_reward_repeat_penalty_std": 0.14054490625858307, "eval_reward_total_composite_mean": 0.5367342164883246, "eval_reward_total_composite_std": 0.3409125472490604} {"timestamp_utc": "2026-04-12T00:24:39Z", "mode": "train", "global_step": 1551, "epoch": 0.06229666224846367, "loss": 0.0162, "grad_norm": 1.714144229888916, "learning_rate": 5.303030303030303e-06, "num_tokens": 3481876.0, "completions/mean_length": 168.125, "completions/min_length": 159.0, "completions/max_length": 174.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 168.125, "completions/min_terminated_length": 159.0, "completions/max_terminated_length": 174.0, "rewards/meter/mean": 0.9984342455863953, "rewards/meter/std": 0.00023047976719681174, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7916666865348816, "rewards/repeat_penalty/std": 0.0927247703075409, "rewards/total_composite/mean": 0.7904214859008789, "rewards/total_composite/std": 0.09253095835447311, "reward": 0.7904214859008789, "reward_std": 0.09253095090389252, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02464020438492298, "sampling/sampling_logp_difference/max": 2.094449281692505, "sampling/importance_sampling_ratio/min": 0.12313804030418396, "sampling/importance_sampling_ratio/mean": 1.002803921699524, "sampling/importance_sampling_ratio/max": 1.999273419380188, "entropy": 0.14000887237489223, "clip_ratio/low_mean": 0.006647313362918794, "clip_ratio/low_min": 0.006647313362918794, "clip_ratio/high_mean": 0.012758038472384214, "clip_ratio/high_max": 0.012758038472384214, "clip_ratio/region_mean": 0.01940535183530301, "reward_total_mean": 0.7904214859008789, "reward_meter_mean": 0.9984342455863953, "reward_meter_std": 0.00023047976719681174, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7916666865348816, "reward_repeat_penalty_std": 0.0927247703075409, "reward_total_composite_mean": 0.7904214859008789, "reward_total_composite_std": 0.09253095835447311} {"timestamp_utc": "2026-04-12T00:24:45Z", "mode": "train", "global_step": 1552, "epoch": 0.06233682773024862, "loss": -0.01, "grad_norm": 2.4818339347839355, "learning_rate": 5.300000000000001e-06, "num_tokens": 3485227.0, "completions/mean_length": 212.875, "completions/min_length": 197.0, "completions/max_length": 223.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 212.875, "completions/min_terminated_length": 197.0, "completions/max_terminated_length": 223.0, "rewards/meter/mean": 0.9701088666915894, "rewards/meter/std": 0.02329552173614502, "rewards/count_adherence/mean": 0.828125, "rewards/count_adherence/std": 0.06469365209341049, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8910256624221802, "rewards/repeat_penalty/std": 0.1113983765244484, "rewards/total_composite/mean": 0.7158641219139099, "rewards/total_composite/std": 0.10737311840057373, "reward": 0.7158641219139099, "reward_std": 0.10737311094999313, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.044115372002124786, "sampling/sampling_logp_difference/max": 1.539175033569336, "sampling/importance_sampling_ratio/min": 0.21455803513526917, "sampling/importance_sampling_ratio/mean": 1.0085928440093994, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4306017607450485, "clip_ratio/low_mean": 0.009651749860495329, "clip_ratio/low_min": 0.009651749860495329, "clip_ratio/high_mean": 0.026524552376940846, "clip_ratio/high_max": 0.026524552376940846, "clip_ratio/region_mean": 0.036176302237436175, "reward_total_mean": 0.7158641219139099, "reward_meter_mean": 0.9701088666915894, "reward_meter_std": 0.02329552173614502, "reward_count_adherence_mean": 0.828125, "reward_count_adherence_std": 0.06469365209341049, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8910256624221802, "reward_repeat_penalty_std": 0.1113983765244484, "reward_total_composite_mean": 0.7158641219139099, "reward_total_composite_std": 0.10737311840057373} {"timestamp_utc": "2026-04-12T00:24:50Z", "mode": "train", "global_step": 1553, "epoch": 0.06237699321203358, "loss": -0.0033, "grad_norm": 7.115604400634766, "learning_rate": 5.296969696969697e-06, "num_tokens": 3487126.0, "completions/mean_length": 65.375, "completions/min_length": 63.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.375, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.977942705154419, "rewards/meter/std": 0.007237757556140423, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.977942705154419, "rewards/total_composite/std": 0.007237757556140423, "reward": 0.977942705154419, "reward_std": 0.007237738464027643, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03986657038331032, "sampling/sampling_logp_difference/max": 0.8986415863037109, "sampling/importance_sampling_ratio/min": 0.40712234377861023, "sampling/importance_sampling_ratio/mean": 1.0108517408370972, "sampling/importance_sampling_ratio/max": 1.691331386566162, "entropy": 0.40439430996775627, "clip_ratio/low_mean": 0.0174350953893736, "clip_ratio/low_min": 0.0174350953893736, "clip_ratio/high_mean": 0.022814685944467783, "clip_ratio/high_max": 0.022814685944467783, "clip_ratio/region_mean": 0.04024978133384138, "reward_total_mean": 0.977942705154419, "reward_meter_mean": 0.977942705154419, "reward_meter_std": 0.007237757556140423, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.977942705154419, "reward_total_composite_std": 0.007237757556140423} {"timestamp_utc": "2026-04-12T00:24:58Z", "mode": "train", "global_step": 1554, "epoch": 0.06241715869381853, "loss": -0.0326, "grad_norm": 2.9659626483917236, "learning_rate": 5.293939393939395e-06, "num_tokens": 3490583.0, "completions/mean_length": 229.125, "completions/min_length": 206.0, "completions/max_length": 240.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 229.125, "completions/min_terminated_length": 206.0, "completions/max_terminated_length": 240.0, "rewards/meter/mean": 0.9962524771690369, "rewards/meter/std": 0.002178115537390113, "rewards/count_adherence/mean": 0.859375, "rewards/count_adherence/std": 0.04419417306780815, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8135302066802979, "rewards/repeat_penalty/std": 0.05918560549616814, "rewards/total_composite/mean": 0.6974078416824341, "rewards/total_composite/std": 0.07142963260412216, "reward": 0.6974078416824341, "reward_std": 0.07142962515354156, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013559546321630478, "sampling/sampling_logp_difference/max": 2.815333366394043, "sampling/importance_sampling_ratio/min": 0.059884749352931976, "sampling/importance_sampling_ratio/mean": 1.000474452972412, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0460505741648376, "clip_ratio/low_mean": 0.0021819902467541397, "clip_ratio/low_min": 0.0021819902467541397, "clip_ratio/high_mean": 0.004815331194549799, "clip_ratio/high_max": 0.004815331194549799, "clip_ratio/region_mean": 0.006997321441303939, "reward_total_mean": 0.6974078416824341, "reward_meter_mean": 0.9962524771690369, "reward_meter_std": 0.002178115537390113, "reward_count_adherence_mean": 0.859375, "reward_count_adherence_std": 0.04419417306780815, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8135302066802979, "reward_repeat_penalty_std": 0.05918560549616814, "reward_total_composite_mean": 0.6974078416824341, "reward_total_composite_std": 0.07142963260412216} {"timestamp_utc": "2026-04-12T00:25:04Z", "mode": "train", "global_step": 1555, "epoch": 0.062457324175603485, "loss": 0.0094, "grad_norm": 4.632443428039551, "learning_rate": 5.290909090909091e-06, "num_tokens": 3493632.0, "completions/mean_length": 193.125, "completions/min_length": 188.0, "completions/max_length": 204.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 193.125, "completions/min_terminated_length": 188.0, "completions/max_terminated_length": 204.0, "rewards/meter/mean": 0.9984503984451294, "rewards/meter/std": 0.00041957091889344156, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8125, "rewards/repeat_penalty/std": 0.08256731927394867, "rewards/total_composite/mean": 0.6953567266464233, "rewards/total_composite/std": 0.0707344338297844, "reward": 0.6953567266464233, "reward_std": 0.0707344338297844, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015556903555989265, "sampling/sampling_logp_difference/max": 1.6991368532180786, "sampling/importance_sampling_ratio/min": 0.18284127116203308, "sampling/importance_sampling_ratio/mean": 0.9994859099388123, "sampling/importance_sampling_ratio/max": 1.899730920791626, "entropy": 0.05383288115262985, "clip_ratio/low_mean": 0.0032157512614503503, "clip_ratio/low_min": 0.0032157512614503503, "clip_ratio/high_mean": 0.00838999031111598, "clip_ratio/high_max": 0.00838999031111598, "clip_ratio/region_mean": 0.01160574157256633, "reward_total_mean": 0.6953567266464233, "reward_meter_mean": 0.9984503984451294, "reward_meter_std": 0.00041957091889344156, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8125, "reward_repeat_penalty_std": 0.08256731927394867, "reward_total_composite_mean": 0.6953567266464233, "reward_total_composite_std": 0.0707344338297844} {"timestamp_utc": "2026-04-12T00:25:09Z", "mode": "train", "global_step": 1556, "epoch": 0.06249748965738844, "loss": 0.0201, "grad_norm": 3.2562649250030518, "learning_rate": 5.287878787878788e-06, "num_tokens": 3495501.0, "completions/mean_length": 71.625, "completions/min_length": 69.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.625, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9960213303565979, "rewards/meter/std": 0.0024177224840968847, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9960213303565979, "rewards/total_composite/std": 0.0024177224840968847, "reward": 0.9960213303565979, "reward_std": 0.0024177224840968847, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02655681222677231, "sampling/sampling_logp_difference/max": 1.301830768585205, "sampling/importance_sampling_ratio/min": 0.27203330397605896, "sampling/importance_sampling_ratio/mean": 1.0080082416534424, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11559192929416895, "clip_ratio/low_mean": 0.010138889076188207, "clip_ratio/low_min": 0.010138889076188207, "clip_ratio/high_mean": 0.014115164172835648, "clip_ratio/high_max": 0.014115164172835648, "clip_ratio/region_mean": 0.024254053249023855, "reward_total_mean": 0.9960213303565979, "reward_meter_mean": 0.9960213303565979, "reward_meter_std": 0.0024177224840968847, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9960213303565979, "reward_total_composite_std": 0.0024177224840968847} {"timestamp_utc": "2026-04-12T00:25:14Z", "mode": "train", "global_step": 1557, "epoch": 0.06253765513917339, "loss": -0.0016, "grad_norm": 1.289771556854248, "learning_rate": 5.284848484848485e-06, "num_tokens": 3497301.0, "completions/mean_length": 68.0, "completions/min_length": 68.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.0, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9974825382232666, "rewards/meter/std": 0.0009680635412223637, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9974825382232666, "rewards/total_composite/std": 0.0009680635412223637, "reward": 0.9974825382232666, "reward_std": 0.000968057313002646, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007437488529831171, "sampling/sampling_logp_difference/max": 0.3382420539855957, "sampling/importance_sampling_ratio/min": 0.7130227088928223, "sampling/importance_sampling_ratio/mean": 1.0034716129302979, "sampling/importance_sampling_ratio/max": 1.1250584125518799, "entropy": 0.06252996809780598, "clip_ratio/low_mean": 0.0018382353009656072, "clip_ratio/low_min": 0.0018382353009656072, "clip_ratio/high_mean": 0.0018382353009656072, "clip_ratio/high_max": 0.0018382353009656072, "clip_ratio/region_mean": 0.0036764706019312143, "reward_total_mean": 0.9974825382232666, "reward_meter_mean": 0.9974825382232666, "reward_meter_std": 0.0009680635412223637, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9974825382232666, "reward_total_composite_std": 0.0009680635412223637} {"timestamp_utc": "2026-04-12T00:25:18Z", "mode": "train", "global_step": 1558, "epoch": 0.06257782062095835, "loss": 0.0063, "grad_norm": 1.997252345085144, "learning_rate": 5.281818181818183e-06, "num_tokens": 3498783.0, "completions/mean_length": 35.25, "completions/min_length": 35.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.25, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9977306127548218, "rewards/meter/std": 4.97752262162976e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977306127548218, "rewards/total_composite/std": 4.97752262162976e-05, "reward": 0.9977306127548218, "reward_std": 4.97752262162976e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00860452838242054, "sampling/sampling_logp_difference/max": 0.33144044876098633, "sampling/importance_sampling_ratio/min": 0.7421995401382446, "sampling/importance_sampling_ratio/mean": 1.0048854351043701, "sampling/importance_sampling_ratio/max": 1.3929730653762817, "entropy": 0.06664726650342345, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.010714285774156451, "clip_ratio/high_max": 0.010714285774156451, "clip_ratio/region_mean": 0.010714285774156451, "reward_total_mean": 0.9977306127548218, "reward_meter_mean": 0.9977306127548218, "reward_meter_std": 4.97752262162976e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9977306127548218, "reward_total_composite_std": 4.97752262162976e-05} {"timestamp_utc": "2026-04-12T00:25:23Z", "mode": "train", "global_step": 1559, "epoch": 0.0626179861027433, "loss": -0.0004, "grad_norm": 2.993905782699585, "learning_rate": 5.278787878787879e-06, "num_tokens": 3500559.0, "completions/mean_length": 62.0, "completions/min_length": 60.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.0, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9898895025253296, "rewards/meter/std": 0.006041497457772493, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9484682083129883, "rewards/total_composite/std": 0.11559666693210602, "reward": 0.9484682083129883, "reward_std": 0.11559665948152542, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0489659458398819, "sampling/sampling_logp_difference/max": 0.9915828704833984, "sampling/importance_sampling_ratio/min": 0.37098902463912964, "sampling/importance_sampling_ratio/mean": 1.0192768573760986, "sampling/importance_sampling_ratio/max": 1.9443280696868896, "entropy": 0.4067396577447653, "clip_ratio/low_mean": 0.002016128972172737, "clip_ratio/low_min": 0.002016128972172737, "clip_ratio/high_mean": 0.0341619870159775, "clip_ratio/high_max": 0.0341619870159775, "clip_ratio/region_mean": 0.03617811598815024, "reward_total_mean": 0.9484682083129883, "reward_meter_mean": 0.9898895025253296, "reward_meter_std": 0.006041497457772493, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.9484682083129883, "reward_total_composite_std": 0.11559666693210602} {"timestamp_utc": "2026-04-12T00:25:29Z", "mode": "train", "global_step": 1560, "epoch": 0.06265815158452825, "loss": 0.0004, "grad_norm": 3.2149598598480225, "learning_rate": 5.2757575757575764e-06, "num_tokens": 3503371.0, "completions/mean_length": 164.5, "completions/min_length": 155.0, "completions/max_length": 177.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 164.5, "completions/min_terminated_length": 155.0, "completions/max_terminated_length": 177.0, "rewards/meter/mean": 0.9614261388778687, "rewards/meter/std": 0.03620919957756996, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333134651184, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.9209266901016235, "rewards/total_composite/std": 0.058246713131666183, "reward": 0.9209266901016235, "reward_std": 0.05824670195579529, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06021958962082863, "sampling/sampling_logp_difference/max": 1.890352487564087, "sampling/importance_sampling_ratio/min": 0.15101857483386993, "sampling/importance_sampling_ratio/mean": 1.010610580444336, "sampling/importance_sampling_ratio/max": 1.8964616060256958, "entropy": 0.6054803691804409, "clip_ratio/low_mean": 0.020772642455995083, "clip_ratio/low_min": 0.020772642455995083, "clip_ratio/high_mean": 0.030295601580291986, "clip_ratio/high_max": 0.030295601580291986, "clip_ratio/region_mean": 0.05106824403628707, "reward_total_mean": 0.9209266901016235, "reward_meter_mean": 0.9614261388778687, "reward_meter_std": 0.03620919957756996, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333134651184, "reward_repeat_penalty_std": 0.05750546231865883, "reward_total_composite_mean": 0.9209266901016235, "reward_total_composite_std": 0.058246713131666183} {"timestamp_utc": "2026-04-12T00:25:35Z", "mode": "train", "global_step": 1561, "epoch": 0.06269831706631321, "loss": -0.0154, "grad_norm": 3.249744176864624, "learning_rate": 5.272727272727273e-06, "num_tokens": 3506277.0, "completions/mean_length": 172.25, "completions/min_length": 161.0, "completions/max_length": 180.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 172.25, "completions/min_terminated_length": 161.0, "completions/max_terminated_length": 180.0, "rewards/meter/mean": 0.9807374477386475, "rewards/meter/std": 0.04032743349671364, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9305555820465088, "rewards/repeat_penalty/std": 0.08266931027173996, "rewards/total_composite/mean": 0.9116497039794922, "rewards/total_composite/std": 0.07851860672235489, "reward": 0.9116497039794922, "reward_std": 0.07851860672235489, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06819657236337662, "sampling/sampling_logp_difference/max": 2.131275177001953, "sampling/importance_sampling_ratio/min": 0.11868584156036377, "sampling/importance_sampling_ratio/mean": 1.0178204774856567, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5600994266569614, "clip_ratio/low_mean": 0.026302700862288475, "clip_ratio/low_min": 0.026302700862288475, "clip_ratio/high_mean": 0.01949126087129116, "clip_ratio/high_max": 0.01949126087129116, "clip_ratio/region_mean": 0.045793961733579636, "reward_total_mean": 0.9116497039794922, "reward_meter_mean": 0.9807374477386475, "reward_meter_std": 0.04032743349671364, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9305555820465088, "reward_repeat_penalty_std": 0.08266931027173996, "reward_total_composite_mean": 0.9116497039794922, "reward_total_composite_std": 0.07851860672235489} {"timestamp_utc": "2026-04-12T00:25:40Z", "mode": "train", "global_step": 1562, "epoch": 0.06273848254809816, "loss": 0.0416, "grad_norm": 7.787070274353027, "learning_rate": 5.26969696969697e-06, "num_tokens": 3507984.0, "completions/mean_length": 66.375, "completions/min_length": 63.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.375, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9753642678260803, "rewards/meter/std": 0.012682100757956505, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9753642678260803, "rewards/total_composite/std": 0.012682100757956505, "reward": 0.9753642678260803, "reward_std": 0.012682083994150162, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05309029668569565, "sampling/sampling_logp_difference/max": 0.7393133640289307, "sampling/importance_sampling_ratio/min": 0.4774416387081146, "sampling/importance_sampling_ratio/mean": 1.020971417427063, "sampling/importance_sampling_ratio/max": 1.8774487972259521, "entropy": 0.4796396978199482, "clip_ratio/low_mean": 0.016098485328257084, "clip_ratio/low_min": 0.016098485328257084, "clip_ratio/high_mean": 0.017139121424406767, "clip_ratio/high_max": 0.017139121424406767, "clip_ratio/region_mean": 0.03323760675266385, "reward_total_mean": 0.9753642678260803, "reward_meter_mean": 0.9753642678260803, "reward_meter_std": 0.012682100757956505, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9753642678260803, "reward_total_composite_std": 0.012682100757956505} {"timestamp_utc": "2026-04-12T00:25:45Z", "mode": "train", "global_step": 1563, "epoch": 0.06277864802988312, "loss": -0.0149, "grad_norm": 2.302091121673584, "learning_rate": 5.2666666666666665e-06, "num_tokens": 3510443.0, "completions/mean_length": 137.375, "completions/min_length": 133.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 137.375, "completions/min_terminated_length": 133.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.9985888004302979, "rewards/meter/std": 0.0002428983716527, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.7311152815818787, "rewards/total_composite/std": 0.05054369568824768, "reward": 0.7311152815818787, "reward_std": 0.05054369196295738, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01936904713511467, "sampling/sampling_logp_difference/max": 0.9639506340026855, "sampling/importance_sampling_ratio/min": 0.3813832104206085, "sampling/importance_sampling_ratio/mean": 1.0065100193023682, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15207071043550968, "clip_ratio/low_mean": 0.012661085231229663, "clip_ratio/low_min": 0.012661085231229663, "clip_ratio/high_mean": 0.0008680555620230734, "clip_ratio/high_max": 0.0008680555620230734, "clip_ratio/region_mean": 0.013529140793252736, "reward_total_mean": 0.7311152815818787, "reward_meter_mean": 0.9985888004302979, "reward_meter_std": 0.0002428983716527, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.7311152815818787, "reward_total_composite_std": 0.05054369568824768} {"timestamp_utc": "2026-04-12T00:25:51Z", "mode": "train", "global_step": 1564, "epoch": 0.06281881351166807, "loss": -0.0126, "grad_norm": 5.161999702453613, "learning_rate": 5.263636363636364e-06, "num_tokens": 3512785.0, "completions/mean_length": 131.75, "completions/min_length": 119.0, "completions/max_length": 139.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.75, "completions/min_terminated_length": 119.0, "completions/max_terminated_length": 139.0, "rewards/meter/mean": 0.718358039855957, "rewards/meter/std": 0.32531455159187317, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.692800760269165, "rewards/total_composite/std": 0.32817986607551575, "reward": 0.692800760269165, "reward_std": 0.32817986607551575, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07563211023807526, "sampling/sampling_logp_difference/max": 1.0854854583740234, "sampling/importance_sampling_ratio/min": 0.3377377986907959, "sampling/importance_sampling_ratio/mean": 1.01358962059021, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7511504292488098, "clip_ratio/low_mean": 0.02909582480788231, "clip_ratio/low_min": 0.02909582480788231, "clip_ratio/high_mean": 0.033232659101486206, "clip_ratio/high_max": 0.033232659101486206, "clip_ratio/region_mean": 0.062328483909368515, "reward_total_mean": 0.692800760269165, "reward_meter_mean": 0.718358039855957, "reward_meter_std": 0.32531455159187317, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.692800760269165, "reward_total_composite_std": 0.32817986607551575} {"timestamp_utc": "2026-04-12T00:25:56Z", "mode": "train", "global_step": 1565, "epoch": 0.06285897899345302, "loss": -0.0189, "grad_norm": 6.407342433929443, "learning_rate": 5.26060606060606e-06, "num_tokens": 3514505.0, "completions/mean_length": 66.0, "completions/min_length": 58.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.6359885931015015, "rewards/meter/std": 0.39930614829063416, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6359885931015015, "rewards/total_composite/std": 0.39930614829063416, "reward": 0.6359885931015015, "reward_std": 0.39930611848831177, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07870090752840042, "sampling/sampling_logp_difference/max": 1.5872907638549805, "sampling/importance_sampling_ratio/min": 0.20447883009910583, "sampling/importance_sampling_ratio/mean": 1.0197231769561768, "sampling/importance_sampling_ratio/max": 1.9221627712249756, "entropy": 0.7624497786164284, "clip_ratio/low_mean": 0.025564054027199745, "clip_ratio/low_min": 0.025564054027199745, "clip_ratio/high_mean": 0.041696065571159124, "clip_ratio/high_max": 0.041696065571159124, "clip_ratio/region_mean": 0.06726011959835887, "reward_total_mean": 0.6359885931015015, "reward_meter_mean": 0.6359885931015015, "reward_meter_std": 0.39930614829063416, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6359885931015015, "reward_total_composite_std": 0.39930614829063416} {"timestamp_utc": "2026-04-12T00:26:01Z", "mode": "train", "global_step": 1566, "epoch": 0.06289914447523798, "loss": -0.0005, "grad_norm": 4.360117435455322, "learning_rate": 5.257575757575758e-06, "num_tokens": 3516356.0, "completions/mean_length": 71.375, "completions/min_length": 65.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.375, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9724572896957397, "rewards/meter/std": 0.05403033271431923, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9724572896957397, "rewards/total_composite/std": 0.05403033271431923, "reward": 0.9724572896957397, "reward_std": 0.054030340164899826, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06206238269805908, "sampling/sampling_logp_difference/max": 1.6887876987457275, "sampling/importance_sampling_ratio/min": 0.18474335968494415, "sampling/importance_sampling_ratio/mean": 1.0121489763259888, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6091113425791264, "clip_ratio/low_mean": 0.0053571430034935474, "clip_ratio/low_min": 0.0053571430034935474, "clip_ratio/high_mean": 0.05206052586436272, "clip_ratio/high_max": 0.05206052586436272, "clip_ratio/region_mean": 0.057417668867856264, "reward_total_mean": 0.9724572896957397, "reward_meter_mean": 0.9724572896957397, "reward_meter_std": 0.05403033271431923, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9724572896957397, "reward_total_composite_std": 0.05403033271431923} {"timestamp_utc": "2026-04-12T00:26:05Z", "mode": "train", "global_step": 1567, "epoch": 0.06293930995702293, "loss": 0.0001, "grad_norm": 0.27344831824302673, "learning_rate": 5.2545454545454555e-06, "num_tokens": 3518144.0, "completions/mean_length": 68.5, "completions/min_length": 68.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.5, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9978020191192627, "rewards/meter/std": 2.9081666070851497e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978020191192627, "rewards/total_composite/std": 2.9081666070851497e-05, "reward": 0.9978020191192627, "reward_std": 2.9075883503537625e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010215016081929207, "sampling/sampling_logp_difference/max": 0.7412929534912109, "sampling/importance_sampling_ratio/min": 0.47649744153022766, "sampling/importance_sampling_ratio/mean": 1.0018134117126465, "sampling/importance_sampling_ratio/max": 1.357559084892273, "entropy": 0.0569656016305089, "clip_ratio/low_mean": 0.007247899193316698, "clip_ratio/low_min": 0.007247899193316698, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.007247899193316698, "reward_total_mean": 0.9978020191192627, "reward_meter_mean": 0.9978020191192627, "reward_meter_std": 2.9081666070851497e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9978020191192627, "reward_total_composite_std": 2.9081666070851497e-05} {"timestamp_utc": "2026-04-12T00:26:10Z", "mode": "train", "global_step": 1568, "epoch": 0.06297947543880789, "loss": -0.0006, "grad_norm": 0.3887041211128235, "learning_rate": 5.251515151515152e-06, "num_tokens": 3520369.0, "completions/mean_length": 100.125, "completions/min_length": 100.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.125, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.9977670311927795, "rewards/meter/std": 4.242736758897081e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977670311927795, "rewards/total_composite/std": 4.242736758897081e-05, "reward": 0.9977670311927795, "reward_std": 4.243615330778994e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007036585360765457, "sampling/sampling_logp_difference/max": 0.741631031036377, "sampling/importance_sampling_ratio/min": 0.47633635997772217, "sampling/importance_sampling_ratio/mean": 1.00063157081604, "sampling/importance_sampling_ratio/max": 1.1633974313735962, "entropy": 0.04269820963963866, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9977670311927795, "reward_meter_mean": 0.9977670311927795, "reward_meter_std": 4.242736758897081e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9977670311927795, "reward_total_composite_std": 4.242736758897081e-05} {"timestamp_utc": "2026-04-12T00:26:15Z", "mode": "train", "global_step": 1569, "epoch": 0.06301964092059284, "loss": 0.0094, "grad_norm": 19.87078857421875, "learning_rate": 5.248484848484849e-06, "num_tokens": 3522282.0, "completions/mean_length": 63.125, "completions/min_length": 62.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.125, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.998197078704834, "rewards/meter/std": 0.00027716331533156335, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998197078704834, "rewards/total_composite/std": 0.00027716331533156335, "reward": 0.998197078704834, "reward_std": 0.00027716669137589633, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010409246198832989, "sampling/sampling_logp_difference/max": 0.8942767381668091, "sampling/importance_sampling_ratio/min": 0.40890324115753174, "sampling/importance_sampling_ratio/mean": 0.9992884993553162, "sampling/importance_sampling_ratio/max": 1.3447296619415283, "entropy": 0.033813337329775095, "clip_ratio/low_mean": 0.002016128972172737, "clip_ratio/low_min": 0.002016128972172737, "clip_ratio/high_mean": 0.001953125, "clip_ratio/high_max": 0.001953125, "clip_ratio/region_mean": 0.003969253972172737, "reward_total_mean": 0.998197078704834, "reward_meter_mean": 0.998197078704834, "reward_meter_std": 0.00027716331533156335, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.998197078704834, "reward_total_composite_std": 0.00027716331533156335} {"timestamp_utc": "2026-04-12T00:26:20Z", "mode": "train", "global_step": 1570, "epoch": 0.0630598064023778, "loss": 0.0363, "grad_norm": 6.926992893218994, "learning_rate": 5.245454545454546e-06, "num_tokens": 3524067.0, "completions/mean_length": 65.125, "completions/min_length": 60.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.125, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.7359851598739624, "rewards/meter/std": 0.4039470851421356, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6135989427566528, "rewards/total_composite/std": 0.46367478370666504, "reward": 0.6135989427566528, "reward_std": 0.46367475390434265, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08760727196931839, "sampling/sampling_logp_difference/max": 1.2285175323486328, "sampling/importance_sampling_ratio/min": 0.29272621870040894, "sampling/importance_sampling_ratio/mean": 1.0079141855239868, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8175845369696617, "clip_ratio/low_mean": 0.028282265178859234, "clip_ratio/low_min": 0.028282265178859234, "clip_ratio/high_mean": 0.04102045390754938, "clip_ratio/high_max": 0.04102045390754938, "clip_ratio/region_mean": 0.06930271908640862, "reward_total_mean": 0.6135989427566528, "reward_meter_mean": 0.7359851598739624, "reward_meter_std": 0.4039470851421356, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6135989427566528, "reward_total_composite_std": 0.46367478370666504} {"timestamp_utc": "2026-04-12T00:26:26Z", "mode": "train", "global_step": 1571, "epoch": 0.06309997188416275, "loss": 0.0237, "grad_norm": 2.949753999710083, "learning_rate": 5.242424242424244e-06, "num_tokens": 3527020.0, "completions/mean_length": 171.125, "completions/min_length": 168.0, "completions/max_length": 185.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 171.125, "completions/min_terminated_length": 168.0, "completions/max_terminated_length": 185.0, "rewards/meter/mean": 0.9974609613418579, "rewards/meter/std": 0.0009353406494483352, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7897727489471436, "rewards/repeat_penalty/std": 0.03682740405201912, "rewards/total_composite/mean": 0.767343282699585, "rewards/total_composite/std": 0.05765299126505852, "reward": 0.767343282699585, "reward_std": 0.05765299126505852, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014616935513913631, "sampling/sampling_logp_difference/max": 1.0565416812896729, "sampling/importance_sampling_ratio/min": 0.46636128425598145, "sampling/importance_sampling_ratio/mean": 0.9989144206047058, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0667440458200872, "clip_ratio/low_mean": 0.006336958380416036, "clip_ratio/low_min": 0.006336958380416036, "clip_ratio/high_mean": 0.011825977824628353, "clip_ratio/high_max": 0.011825977824628353, "clip_ratio/region_mean": 0.01816293620504439, "reward_total_mean": 0.767343282699585, "reward_meter_mean": 0.9974609613418579, "reward_meter_std": 0.0009353406494483352, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7897727489471436, "reward_repeat_penalty_std": 0.03682740405201912, "reward_total_composite_mean": 0.767343282699585, "reward_total_composite_std": 0.05765299126505852} {"timestamp_utc": "2026-04-12T00:26:36Z", "mode": "train", "global_step": 1572, "epoch": 0.0631401373659477, "loss": -0.218, "grad_norm": 1.639418125152588, "learning_rate": 5.23939393939394e-06, "num_tokens": 3530638.0, "completions/mean_length": 308.25, "completions/min_length": 265.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 279.14288330078125, "completions/min_terminated_length": 265.0, "completions/max_terminated_length": 297.0, "rewards/meter/mean": 0.6877518892288208, "rewards/meter/std": 0.3425425887107849, "rewards/count_adherence/mean": 0.862500011920929, "rewards/count_adherence/std": 0.0517548993229866, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.9172793626785278, "rewards/repeat_penalty/std": 0.07193652540445328, "rewards/total_composite/mean": 0.5252456665039062, "rewards/total_composite/std": 0.28777360916137695, "reward": 0.5252456665039062, "reward_std": 0.28777357935905457, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0495113804936409, "sampling/sampling_logp_difference/max": 2.470219373703003, "sampling/importance_sampling_ratio/min": 0.08456631004810333, "sampling/importance_sampling_ratio/mean": 1.012639045715332, "sampling/importance_sampling_ratio/max": 1.8714102506637573, "entropy": 0.41537686437368393, "clip_ratio/low_mean": 0.015551732387393713, "clip_ratio/low_min": 0.015551732387393713, "clip_ratio/high_mean": 0.012186205829493701, "clip_ratio/high_max": 0.012186205829493701, "clip_ratio/region_mean": 0.027737938216887414, "reward_total_mean": 0.5252456665039062, "reward_meter_mean": 0.6877518892288208, "reward_meter_std": 0.3425425887107849, "reward_count_adherence_mean": 0.862500011920929, "reward_count_adherence_std": 0.0517548993229866, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.9172793626785278, "reward_repeat_penalty_std": 0.07193652540445328, "reward_total_composite_mean": 0.5252456665039062, "reward_total_composite_std": 0.28777360916137695} {"timestamp_utc": "2026-04-12T00:26:41Z", "mode": "train", "global_step": 1573, "epoch": 0.06318030284773266, "loss": 0.0007, "grad_norm": 5.333371162414551, "learning_rate": 5.236363636363637e-06, "num_tokens": 3532465.0, "completions/mean_length": 64.375, "completions/min_length": 61.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.375, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.6559607982635498, "rewards/meter/std": 0.3895518183708191, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6559607982635498, "rewards/total_composite/std": 0.3895518183708191, "reward": 0.6559607982635498, "reward_std": 0.3895518183708191, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07227914035320282, "sampling/sampling_logp_difference/max": 1.8713045120239258, "sampling/importance_sampling_ratio/min": 0.15392273664474487, "sampling/importance_sampling_ratio/mean": 1.0150277614593506, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7125434167683125, "clip_ratio/low_mean": 0.033473065122962, "clip_ratio/low_min": 0.033473065122962, "clip_ratio/high_mean": 0.047869643196463585, "clip_ratio/high_max": 0.047869643196463585, "clip_ratio/region_mean": 0.08134270831942558, "reward_total_mean": 0.6559607982635498, "reward_meter_mean": 0.6559607982635498, "reward_meter_std": 0.3895518183708191, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6559607982635498, "reward_total_composite_std": 0.3895518183708191} {"timestamp_utc": "2026-04-12T00:26:46Z", "mode": "train", "global_step": 1574, "epoch": 0.06322046832951761, "loss": -0.0041, "grad_norm": 6.877980709075928, "learning_rate": 5.233333333333334e-06, "num_tokens": 3534610.0, "completions/mean_length": 94.125, "completions/min_length": 88.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.125, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.8490567207336426, "rewards/meter/std": 0.342186838388443, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8243577480316162, "rewards/total_composite/std": 0.3378320336341858, "reward": 0.8243577480316162, "reward_std": 0.3378320336341858, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0785253494977951, "sampling/sampling_logp_difference/max": 2.2510461807250977, "sampling/importance_sampling_ratio/min": 0.10528901219367981, "sampling/importance_sampling_ratio/mean": 1.00845468044281, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8282850012183189, "clip_ratio/low_mean": 0.00818392145447433, "clip_ratio/low_min": 0.00818392145447433, "clip_ratio/high_mean": 0.04302619933150709, "clip_ratio/high_max": 0.04302619933150709, "clip_ratio/region_mean": 0.05121012078598142, "reward_total_mean": 0.8243577480316162, "reward_meter_mean": 0.8490567207336426, "reward_meter_std": 0.342186838388443, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.8243577480316162, "reward_total_composite_std": 0.3378320336341858} {"timestamp_utc": "2026-04-12T00:26:51Z", "mode": "train", "global_step": 1575, "epoch": 0.06326063381130256, "loss": -0.0016, "grad_norm": 6.546207427978516, "learning_rate": 5.230303030303031e-06, "num_tokens": 3536318.0, "completions/mean_length": 61.5, "completions/min_length": 59.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.5, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9349583387374878, "rewards/meter/std": 0.10497313737869263, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9349583387374878, "rewards/total_composite/std": 0.10497313737869263, "reward": 0.9349583387374878, "reward_std": 0.10497313737869263, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04767481982707977, "sampling/sampling_logp_difference/max": 1.3624334335327148, "sampling/importance_sampling_ratio/min": 0.2560369670391083, "sampling/importance_sampling_ratio/mean": 1.0037705898284912, "sampling/importance_sampling_ratio/max": 1.711548924446106, "entropy": 0.3381665423512459, "clip_ratio/low_mean": 0.016170635353773832, "clip_ratio/low_min": 0.016170635353773832, "clip_ratio/high_mean": 0.02857396099716425, "clip_ratio/high_max": 0.02857396099716425, "clip_ratio/region_mean": 0.04474459635093808, "reward_total_mean": 0.9349583387374878, "reward_meter_mean": 0.9349583387374878, "reward_meter_std": 0.10497313737869263, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9349583387374878, "reward_total_composite_std": 0.10497313737869263} {"timestamp_utc": "2026-04-12T00:26:57Z", "mode": "train", "global_step": 1576, "epoch": 0.06330079929308752, "loss": 0.0111, "grad_norm": 3.9961159229278564, "learning_rate": 5.2272727272727274e-06, "num_tokens": 3538180.0, "completions/mean_length": 61.75, "completions/min_length": 58.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.75, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9147671461105347, "rewards/meter/std": 0.21282289922237396, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9147671461105347, "rewards/total_composite/std": 0.21282289922237396, "reward": 0.9147671461105347, "reward_std": 0.21282285451889038, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05491922050714493, "sampling/sampling_logp_difference/max": 1.6048030853271484, "sampling/importance_sampling_ratio/min": 0.20092912018299103, "sampling/importance_sampling_ratio/mean": 1.0070737600326538, "sampling/importance_sampling_ratio/max": 1.9876987934112549, "entropy": 0.42180035449564457, "clip_ratio/low_mean": 0.001953125, "clip_ratio/low_min": 0.001953125, "clip_ratio/high_mean": 0.03661064524203539, "clip_ratio/high_max": 0.03661064524203539, "clip_ratio/region_mean": 0.03856377024203539, "reward_total_mean": 0.9147671461105347, "reward_meter_mean": 0.9147671461105347, "reward_meter_std": 0.21282289922237396, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9147671461105347, "reward_total_composite_std": 0.21282289922237396} {"timestamp_utc": "2026-04-12T00:27:02Z", "mode": "train", "global_step": 1577, "epoch": 0.06334096477487247, "loss": 0.0008, "grad_norm": 4.388180732727051, "learning_rate": 5.224242424242425e-06, "num_tokens": 3540188.0, "completions/mean_length": 92.0, "completions/min_length": 89.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.0, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.9917011260986328, "rewards/meter/std": 0.006606912240386009, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8250000476837158, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8181818723678589, "rewards/total_composite/std": 0.07076939195394516, "reward": 0.8181818723678589, "reward_std": 0.07076937705278397, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.037422213703393936, "sampling/sampling_logp_difference/max": 1.1900215148925781, "sampling/importance_sampling_ratio/min": 0.3042147159576416, "sampling/importance_sampling_ratio/mean": 1.0047838687896729, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.24610379338264465, "clip_ratio/low_mean": 0.02302604599390179, "clip_ratio/low_min": 0.02302604599390179, "clip_ratio/high_mean": 0.005319148767739534, "clip_ratio/high_max": 0.005319148767739534, "clip_ratio/region_mean": 0.028345194761641324, "reward_total_mean": 0.8181818723678589, "reward_meter_mean": 0.9917011260986328, "reward_meter_std": 0.006606912240386009, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8250000476837158, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.8181818723678589, "reward_total_composite_std": 0.07076939195394516} {"timestamp_utc": "2026-04-12T00:27:09Z", "mode": "train", "global_step": 1578, "epoch": 0.06338113025665743, "loss": 0.0638, "grad_norm": 4.709424018859863, "learning_rate": 5.221212121212121e-06, "num_tokens": 3543292.0, "completions/mean_length": 191.0, "completions/min_length": 175.0, "completions/max_length": 218.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 191.0, "completions/min_terminated_length": 175.0, "completions/max_terminated_length": 218.0, "rewards/meter/mean": 0.9755741953849792, "rewards/meter/std": 0.027564238756895065, "rewards/count_adherence/mean": 0.925000011920929, "rewards/count_adherence/std": 0.1035098284482956, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9154040217399597, "rewards/repeat_penalty/std": 0.10342530906200409, "rewards/total_composite/mean": 0.8337088823318481, "rewards/total_composite/std": 0.17133085429668427, "reward": 0.8337088823318481, "reward_std": 0.17133086919784546, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05211658403277397, "sampling/sampling_logp_difference/max": 2.6808853149414062, "sampling/importance_sampling_ratio/min": 0.06850247830152512, "sampling/importance_sampling_ratio/mean": 1.0094478130340576, "sampling/importance_sampling_ratio/max": 1.927268385887146, "entropy": 0.4086650311946869, "clip_ratio/low_mean": 0.012475505471229553, "clip_ratio/low_min": 0.012475505471229553, "clip_ratio/high_mean": 0.02987705054692924, "clip_ratio/high_max": 0.02987705054692924, "clip_ratio/region_mean": 0.04235255601815879, "reward_total_mean": 0.8337088823318481, "reward_meter_mean": 0.9755741953849792, "reward_meter_std": 0.027564238756895065, "reward_count_adherence_mean": 0.925000011920929, "reward_count_adherence_std": 0.1035098284482956, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9154040217399597, "reward_repeat_penalty_std": 0.10342530906200409, "reward_total_composite_mean": 0.8337088823318481, "reward_total_composite_std": 0.17133085429668427} {"timestamp_utc": "2026-04-12T00:27:14Z", "mode": "train", "global_step": 1579, "epoch": 0.06342129573844238, "loss": 0.008, "grad_norm": 2.8309895992279053, "learning_rate": 5.218181818181819e-06, "num_tokens": 3545353.0, "completions/mean_length": 100.625, "completions/min_length": 100.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.625, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.9977493286132812, "rewards/meter/std": 0.00013334653340280056, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.9478673934936523, "rewards/total_composite/std": 0.09243174642324448, "reward": 0.9478673934936523, "reward_std": 0.09243172407150269, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007003183010965586, "sampling/sampling_logp_difference/max": 0.8175020217895508, "sampling/importance_sampling_ratio/min": 0.5253335237503052, "sampling/importance_sampling_ratio/mean": 1.0029453039169312, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.042127607855945826, "clip_ratio/low_mean": 0.003676470718346536, "clip_ratio/low_min": 0.003676470718346536, "clip_ratio/high_mean": 0.0012376237427815795, "clip_ratio/high_max": 0.0012376237427815795, "clip_ratio/region_mean": 0.004914094461128116, "reward_total_mean": 0.9478673934936523, "reward_meter_mean": 0.9977493286132812, "reward_meter_std": 0.00013334653340280056, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.9478673934936523, "reward_total_composite_std": 0.09243174642324448} {"timestamp_utc": "2026-04-12T00:27:21Z", "mode": "train", "global_step": 1580, "epoch": 0.06346146122022733, "loss": 0.017, "grad_norm": 3.488912343978882, "learning_rate": 5.215151515151516e-06, "num_tokens": 3548859.0, "completions/mean_length": 233.25, "completions/min_length": 217.0, "completions/max_length": 243.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 233.25, "completions/min_terminated_length": 217.0, "completions/max_terminated_length": 243.0, "rewards/meter/mean": 0.6562398672103882, "rewards/meter/std": 0.2839363217353821, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9615384340286255, "rewards/repeat_penalty/std": 0.082234226167202, "rewards/total_composite/mean": 0.6222666501998901, "rewards/total_composite/std": 0.2602792978286743, "reward": 0.6222666501998901, "reward_std": 0.2602792978286743, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09925277531147003, "sampling/sampling_logp_difference/max": 1.4719266891479492, "sampling/importance_sampling_ratio/min": 0.22948291897773743, "sampling/importance_sampling_ratio/mean": 1.0219330787658691, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0062834583222866, "clip_ratio/low_mean": 0.028440887574106455, "clip_ratio/low_min": 0.028440887574106455, "clip_ratio/high_mean": 0.03686255170032382, "clip_ratio/high_max": 0.03686255170032382, "clip_ratio/region_mean": 0.06530343927443027, "reward_total_mean": 0.6222666501998901, "reward_meter_mean": 0.6562398672103882, "reward_meter_std": 0.2839363217353821, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9615384340286255, "reward_repeat_penalty_std": 0.082234226167202, "reward_total_composite_mean": 0.6222666501998901, "reward_total_composite_std": 0.2602792978286743} {"timestamp_utc": "2026-04-12T00:27:28Z", "mode": "train", "global_step": 1581, "epoch": 0.06350162670201229, "loss": -0.0056, "grad_norm": 1.136604905128479, "learning_rate": 5.212121212121213e-06, "num_tokens": 3552445.0, "completions/mean_length": 250.25, "completions/min_length": 240.0, "completions/max_length": 255.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 250.25, "completions/min_terminated_length": 240.0, "completions/max_terminated_length": 255.0, "rewards/meter/mean": 0.9991568922996521, "rewards/meter/std": 0.00013457036402542144, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7211538553237915, "rewards/repeat_penalty/std": 0.039811473339796066, "rewards/total_composite/mean": 0.7205430865287781, "rewards/total_composite/std": 0.039719030261039734, "reward": 0.7205430865287781, "reward_std": 0.03971904516220093, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017896855250000954, "sampling/sampling_logp_difference/max": 1.916792631149292, "sampling/importance_sampling_ratio/min": 0.14707794785499573, "sampling/importance_sampling_ratio/mean": 1.0050798654556274, "sampling/importance_sampling_ratio/max": 1.794761300086975, "entropy": 0.12574958708137274, "clip_ratio/low_mean": 0.00662941113114357, "clip_ratio/low_min": 0.00662941113114357, "clip_ratio/high_mean": 0.0039564905455335975, "clip_ratio/high_max": 0.0039564905455335975, "clip_ratio/region_mean": 0.010585901676677167, "reward_total_mean": 0.7205430865287781, "reward_meter_mean": 0.9991568922996521, "reward_meter_std": 0.00013457036402542144, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7211538553237915, "reward_repeat_penalty_std": 0.039811473339796066, "reward_total_composite_mean": 0.7205430865287781, "reward_total_composite_std": 0.039719030261039734} {"timestamp_utc": "2026-04-12T00:27:33Z", "mode": "train", "global_step": 1582, "epoch": 0.06354179218379724, "loss": 0.0002, "grad_norm": 4.802741527557373, "learning_rate": 5.209090909090909e-06, "num_tokens": 3554625.0, "completions/mean_length": 107.5, "completions/min_length": 103.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.5, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.989787757396698, "rewards/meter/std": 0.00834440253674984, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.989787757396698, "rewards/total_composite/std": 0.00834440253674984, "reward": 0.989787757396698, "reward_std": 0.008344405330717564, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05708528682589531, "sampling/sampling_logp_difference/max": 1.5647554397583008, "sampling/importance_sampling_ratio/min": 0.24133498966693878, "sampling/importance_sampling_ratio/mean": 1.015655517578125, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4601043425500393, "clip_ratio/low_mean": 0.010625278111547232, "clip_ratio/low_min": 0.010625278111547232, "clip_ratio/high_mean": 0.034577128011733294, "clip_ratio/high_max": 0.034577128011733294, "clip_ratio/region_mean": 0.045202406123280525, "reward_total_mean": 0.989787757396698, "reward_meter_mean": 0.989787757396698, "reward_meter_std": 0.00834440253674984, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.989787757396698, "reward_total_composite_std": 0.00834440253674984} {"timestamp_utc": "2026-04-12T00:27:38Z", "mode": "train", "global_step": 1583, "epoch": 0.0635819576655822, "loss": 0.0415, "grad_norm": 13.880090713500977, "learning_rate": 5.2060606060606065e-06, "num_tokens": 3556523.0, "completions/mean_length": 67.25, "completions/min_length": 64.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.25, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.7919849157333374, "rewards/meter/std": 0.3320651650428772, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7919849157333374, "rewards/total_composite/std": 0.3320651650428772, "reward": 0.7919849157333374, "reward_std": 0.3320651352405548, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.060586389154195786, "sampling/sampling_logp_difference/max": 1.4995098114013672, "sampling/importance_sampling_ratio/min": 0.22323958575725555, "sampling/importance_sampling_ratio/mean": 1.0017415285110474, "sampling/importance_sampling_ratio/max": 1.9709203243255615, "entropy": 0.5009849853813648, "clip_ratio/low_mean": 0.010714286006987095, "clip_ratio/low_min": 0.010714286006987095, "clip_ratio/high_mean": 0.03560801479034126, "clip_ratio/high_max": 0.03560801479034126, "clip_ratio/region_mean": 0.04632230079732835, "reward_total_mean": 0.7919849157333374, "reward_meter_mean": 0.7919849157333374, "reward_meter_std": 0.3320651650428772, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7919849157333374, "reward_total_composite_std": 0.3320651650428772} {"timestamp_utc": "2026-04-12T00:27:43Z", "mode": "train", "global_step": 1584, "epoch": 0.06362212314736715, "loss": 0.0011, "grad_norm": 4.862657070159912, "learning_rate": 5.203030303030303e-06, "num_tokens": 3558329.0, "completions/mean_length": 60.75, "completions/min_length": 58.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.75, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9940457940101624, "rewards/meter/std": 0.0014534186339005828, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9940457940101624, "rewards/total_composite/std": 0.0014534186339005828, "reward": 0.9940457940101624, "reward_std": 0.0014534194488078356, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02881665527820587, "sampling/sampling_logp_difference/max": 1.269845962524414, "sampling/importance_sampling_ratio/min": 0.28087490797042847, "sampling/importance_sampling_ratio/mean": 1.0062899589538574, "sampling/importance_sampling_ratio/max": 1.5038725137710571, "entropy": 0.22835318371653557, "clip_ratio/low_mean": 0.006222632946446538, "clip_ratio/low_min": 0.006222632946446538, "clip_ratio/high_mean": 0.010113696102052927, "clip_ratio/high_max": 0.010113696102052927, "clip_ratio/region_mean": 0.016336329048499465, "reward_total_mean": 0.9940457940101624, "reward_meter_mean": 0.9940457940101624, "reward_meter_std": 0.0014534186339005828, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9940457940101624, "reward_total_composite_std": 0.0014534186339005828} {"timestamp_utc": "2026-04-12T00:27:48Z", "mode": "train", "global_step": 1585, "epoch": 0.0636622886291521, "loss": 0.0019, "grad_norm": 2.1006689071655273, "learning_rate": 5.2e-06, "num_tokens": 3560777.0, "completions/mean_length": 144.0, "completions/min_length": 144.0, "completions/max_length": 144.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 144.0, "completions/min_terminated_length": 144.0, "completions/max_terminated_length": 144.0, "rewards/meter/mean": 0.9989944696426392, "rewards/meter/std": 1.850287117122207e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8392857313156128, "rewards/repeat_penalty/std": 0.09155284613370895, "rewards/total_composite/mean": 0.8384406566619873, "rewards/total_composite/std": 0.09144823998212814, "reward": 0.8384406566619873, "reward_std": 0.09144823998212814, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010658186860382557, "sampling/sampling_logp_difference/max": 0.8828556537628174, "sampling/importance_sampling_ratio/min": 0.4136001169681549, "sampling/importance_sampling_ratio/mean": 1.0028645992279053, "sampling/importance_sampling_ratio/max": 1.262144923210144, "entropy": 0.0726482942700386, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.007812500116415322, "clip_ratio/high_max": 0.007812500116415322, "clip_ratio/region_mean": 0.007812500116415322, "reward_total_mean": 0.8384406566619873, "reward_meter_mean": 0.9989944696426392, "reward_meter_std": 1.850287117122207e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8392857313156128, "reward_repeat_penalty_std": 0.09155284613370895, "reward_total_composite_mean": 0.8384406566619873, "reward_total_composite_std": 0.09144823998212814} {"timestamp_utc": "2026-04-12T00:27:53Z", "mode": "train", "global_step": 1586, "epoch": 0.06370245411093706, "loss": 0.0059, "grad_norm": 6.0716328620910645, "learning_rate": 5.196969696969697e-06, "num_tokens": 3562502.0, "completions/mean_length": 66.625, "completions/min_length": 65.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.625, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.8443154096603394, "rewards/meter/std": 0.19778741896152496, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8443154096603394, "rewards/total_composite/std": 0.19778741896152496, "reward": 0.8443154096603394, "reward_std": 0.19778741896152496, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0697539895772934, "sampling/sampling_logp_difference/max": 1.1605567932128906, "sampling/importance_sampling_ratio/min": 0.3133116662502289, "sampling/importance_sampling_ratio/mean": 1.0118776559829712, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6144770756363869, "clip_ratio/low_mean": 0.017196210101246834, "clip_ratio/low_min": 0.017196210101246834, "clip_ratio/high_mean": 0.03553921659477055, "clip_ratio/high_max": 0.03553921659477055, "clip_ratio/region_mean": 0.052735426696017385, "reward_total_mean": 0.8443154096603394, "reward_meter_mean": 0.8443154096603394, "reward_meter_std": 0.19778741896152496, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8443154096603394, "reward_total_composite_std": 0.19778741896152496} {"timestamp_utc": "2026-04-12T00:27:58Z", "mode": "train", "global_step": 1587, "epoch": 0.06374261959272201, "loss": -0.0017, "grad_norm": 5.30000114440918, "learning_rate": 5.193939393939395e-06, "num_tokens": 3564369.0, "completions/mean_length": 66.375, "completions/min_length": 64.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.375, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9182997941970825, "rewards/meter/std": 0.08735460042953491, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9182997941970825, "rewards/total_composite/std": 0.08735460042953491, "reward": 0.9182997941970825, "reward_std": 0.08735460788011551, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06299924105405807, "sampling/sampling_logp_difference/max": 1.8149499893188477, "sampling/importance_sampling_ratio/min": 0.16284604370594025, "sampling/importance_sampling_ratio/mean": 1.0096309185028076, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5464729219675064, "clip_ratio/low_mean": 0.01317360415123403, "clip_ratio/low_min": 0.01317360415123403, "clip_ratio/high_mean": 0.03026041854172945, "clip_ratio/high_max": 0.03026041854172945, "clip_ratio/region_mean": 0.04343402269296348, "reward_total_mean": 0.9182997941970825, "reward_meter_mean": 0.9182997941970825, "reward_meter_std": 0.08735460042953491, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9182997941970825, "reward_total_composite_std": 0.08735460042953491} {"timestamp_utc": "2026-04-12T00:28:03Z", "mode": "train", "global_step": 1588, "epoch": 0.06378278507450696, "loss": 0.0024, "grad_norm": 5.146764755249023, "learning_rate": 5.190909090909091e-06, "num_tokens": 3566296.0, "completions/mean_length": 69.875, "completions/min_length": 66.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.875, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.6954999566078186, "rewards/meter/std": 0.4150845408439636, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6954999566078186, "rewards/total_composite/std": 0.4150845408439636, "reward": 0.6954999566078186, "reward_std": 0.41508448123931885, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0591333732008934, "sampling/sampling_logp_difference/max": 1.3553905487060547, "sampling/importance_sampling_ratio/min": 0.2578465938568115, "sampling/importance_sampling_ratio/mean": 1.014699101448059, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4779812954366207, "clip_ratio/low_mean": 0.007130191195756197, "clip_ratio/low_min": 0.007130191195756197, "clip_ratio/high_mean": 0.025014446233399212, "clip_ratio/high_max": 0.025014446233399212, "clip_ratio/region_mean": 0.03214463742915541, "reward_total_mean": 0.6954999566078186, "reward_meter_mean": 0.6954999566078186, "reward_meter_std": 0.4150845408439636, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6954999566078186, "reward_total_composite_std": 0.4150845408439636} {"timestamp_utc": "2026-04-12T00:28:08Z", "mode": "train", "global_step": 1589, "epoch": 0.06382295055629192, "loss": -0.0052, "grad_norm": 5.699836730957031, "learning_rate": 5.187878787878788e-06, "num_tokens": 3568219.0, "completions/mean_length": 71.375, "completions/min_length": 69.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.375, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.8743665218353271, "rewards/meter/std": 0.338323175907135, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8743665218353271, "rewards/total_composite/std": 0.338323175907135, "reward": 0.8743665218353271, "reward_std": 0.3383232057094574, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.060609057545661926, "sampling/sampling_logp_difference/max": 1.2876157760620117, "sampling/importance_sampling_ratio/min": 0.275927871465683, "sampling/importance_sampling_ratio/mean": 1.0109214782714844, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4802176021039486, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.04341738054063171, "clip_ratio/high_max": 0.04341738054063171, "clip_ratio/region_mean": 0.04341738054063171, "reward_total_mean": 0.8743665218353271, "reward_meter_mean": 0.8743665218353271, "reward_meter_std": 0.338323175907135, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8743665218353271, "reward_total_composite_std": 0.338323175907135} {"timestamp_utc": "2026-04-12T00:28:15Z", "mode": "train", "global_step": 1590, "epoch": 0.06386311603807687, "loss": -0.0196, "grad_norm": 1.8666749000549316, "learning_rate": 5.184848484848485e-06, "num_tokens": 3571993.0, "completions/mean_length": 271.75, "completions/min_length": 248.0, "completions/max_length": 287.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 271.75, "completions/min_terminated_length": 248.0, "completions/max_terminated_length": 287.0, "rewards/meter/mean": 0.9924838542938232, "rewards/meter/std": 0.004395514260977507, "rewards/count_adherence/mean": 0.8611111640930176, "rewards/count_adherence/std": 0.05143444612622261, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6185267567634583, "rewards/repeat_penalty/std": 0.05406184867024422, "rewards/total_composite/mean": 0.5278421640396118, "rewards/total_composite/std": 0.04846978932619095, "reward": 0.5278421640396118, "reward_std": 0.048469796776771545, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018268149346113205, "sampling/sampling_logp_difference/max": 5.459444046020508, "sampling/importance_sampling_ratio/min": 0.004255921114236116, "sampling/importance_sampling_ratio/mean": 1.0014569759368896, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08188144443556666, "clip_ratio/low_mean": 0.007355856709182262, "clip_ratio/low_min": 0.007355856709182262, "clip_ratio/high_mean": 0.004474195709917694, "clip_ratio/high_max": 0.004474195709917694, "clip_ratio/region_mean": 0.011830052419099957, "reward_total_mean": 0.5278421640396118, "reward_meter_mean": 0.9924838542938232, "reward_meter_std": 0.004395514260977507, "reward_count_adherence_mean": 0.8611111640930176, "reward_count_adherence_std": 0.05143444612622261, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6185267567634583, "reward_repeat_penalty_std": 0.05406184867024422, "reward_total_composite_mean": 0.5278421640396118, "reward_total_composite_std": 0.04846978932619095} {"timestamp_utc": "2026-04-12T00:28:19Z", "mode": "train", "global_step": 1591, "epoch": 0.06390328151986183, "loss": 0.0162, "grad_norm": 6.771106719970703, "learning_rate": 5.181818181818182e-06, "num_tokens": 3573831.0, "completions/mean_length": 67.75, "completions/min_length": 66.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.75, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.89401775598526, "rewards/meter/std": 0.16673576831817627, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.89401775598526, "rewards/total_composite/std": 0.16673576831817627, "reward": 0.89401775598526, "reward_std": 0.16673575341701508, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05256752669811249, "sampling/sampling_logp_difference/max": 1.1782063245773315, "sampling/importance_sampling_ratio/min": 0.40265539288520813, "sampling/importance_sampling_ratio/mean": 1.0108517408370972, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.48275283351540565, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.03321157908067107, "clip_ratio/high_max": 0.03321157908067107, "clip_ratio/region_mean": 0.03321157908067107, "reward_total_mean": 0.89401775598526, "reward_meter_mean": 0.89401775598526, "reward_meter_std": 0.16673576831817627, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.89401775598526, "reward_total_composite_std": 0.16673576831817627} {"timestamp_utc": "2026-04-12T00:28:24Z", "mode": "train", "global_step": 1592, "epoch": 0.06394344700164678, "loss": 0.0026, "grad_norm": 1.81292724609375, "learning_rate": 5.1787878787878784e-06, "num_tokens": 3575592.0, "completions/mean_length": 58.125, "completions/min_length": 58.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.125, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9933634400367737, "rewards/meter/std": 0.00030929798958823085, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9933634400367737, "rewards/total_composite/std": 0.00030929798958823085, "reward": 0.9933634400367737, "reward_std": 0.00030929522472433746, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004956495948135853, "sampling/sampling_logp_difference/max": 0.7400164604187012, "sampling/importance_sampling_ratio/min": 0.4771060645580292, "sampling/importance_sampling_ratio/mean": 0.9998379349708557, "sampling/importance_sampling_ratio/max": 1.1543108224868774, "entropy": 0.020813994109630585, "clip_ratio/low_mean": 0.0021186440717428923, "clip_ratio/low_min": 0.0021186440717428923, "clip_ratio/high_mean": 0.006465517217293382, "clip_ratio/high_max": 0.006465517217293382, "clip_ratio/region_mean": 0.008584161289036274, "reward_total_mean": 0.9933634400367737, "reward_meter_mean": 0.9933634400367737, "reward_meter_std": 0.00030929798958823085, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9933634400367737, "reward_total_composite_std": 0.00030929798958823085} {"timestamp_utc": "2026-04-12T00:28:29Z", "mode": "train", "global_step": 1593, "epoch": 0.06398361248343173, "loss": 0.035, "grad_norm": 10.237826347351074, "learning_rate": 5.1757575757575765e-06, "num_tokens": 3577358.0, "completions/mean_length": 68.75, "completions/min_length": 67.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.75, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.7235053181648254, "rewards/meter/std": 0.3811579644680023, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7235053181648254, "rewards/total_composite/std": 0.3811579644680023, "reward": 0.7235053181648254, "reward_std": 0.3811579942703247, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027932489290833473, "sampling/sampling_logp_difference/max": 1.7486402988433838, "sampling/importance_sampling_ratio/min": 0.17401038110256195, "sampling/importance_sampling_ratio/mean": 0.9999585151672363, "sampling/importance_sampling_ratio/max": 1.7127673625946045, "entropy": 0.12029994325712323, "clip_ratio/low_mean": 0.012177230441011488, "clip_ratio/low_min": 0.012177230441011488, "clip_ratio/high_mean": 0.0018656715983524919, "clip_ratio/high_max": 0.0018656715983524919, "clip_ratio/region_mean": 0.01404290203936398, "reward_total_mean": 0.7235053181648254, "reward_meter_mean": 0.7235053181648254, "reward_meter_std": 0.3811579644680023, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7235053181648254, "reward_total_composite_std": 0.3811579644680023} {"timestamp_utc": "2026-04-12T00:28:36Z", "mode": "train", "global_step": 1594, "epoch": 0.06402377796521669, "loss": 0.0095, "grad_norm": 1.7681658267974854, "learning_rate": 5.172727272727273e-06, "num_tokens": 3580425.0, "completions/mean_length": 195.375, "completions/min_length": 188.0, "completions/max_length": 200.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 195.375, "completions/min_terminated_length": 188.0, "completions/max_terminated_length": 200.0, "rewards/meter/mean": 0.9953435659408569, "rewards/meter/std": 0.0010429132962599397, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7424242496490479, "rewards/repeat_penalty/std": 0.04169124737381935, "rewards/total_composite/mean": 0.7389621734619141, "rewards/total_composite/std": 0.04144872725009918, "reward": 0.7389621734619141, "reward_std": 0.041448723524808884, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016124166548252106, "sampling/sampling_logp_difference/max": 2.298177719116211, "sampling/importance_sampling_ratio/min": 0.10044170916080475, "sampling/importance_sampling_ratio/mean": 0.9993682503700256, "sampling/importance_sampling_ratio/max": 1.846528172492981, "entropy": 0.06844896962866187, "clip_ratio/low_mean": 0.005208890186622739, "clip_ratio/low_min": 0.005208890186622739, "clip_ratio/high_mean": 0.007002223108429462, "clip_ratio/high_max": 0.007002223108429462, "clip_ratio/region_mean": 0.0122111132950522, "reward_total_mean": 0.7389621734619141, "reward_meter_mean": 0.9953435659408569, "reward_meter_std": 0.0010429132962599397, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7424242496490479, "reward_repeat_penalty_std": 0.04169124737381935, "reward_total_composite_mean": 0.7389621734619141, "reward_total_composite_std": 0.04144872725009918} {"timestamp_utc": "2026-04-12T00:28:42Z", "mode": "train", "global_step": 1595, "epoch": 0.06406394344700164, "loss": 0.0842, "grad_norm": 4.416578769683838, "learning_rate": 5.16969696969697e-06, "num_tokens": 3583190.0, "completions/mean_length": 161.625, "completions/min_length": 144.0, "completions/max_length": 194.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 161.625, "completions/min_terminated_length": 144.0, "completions/max_terminated_length": 194.0, "rewards/meter/mean": 0.9966393113136292, "rewards/meter/std": 0.0009264187538065016, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9503968358039856, "rewards/repeat_penalty/std": 0.06915634870529175, "rewards/total_composite/mean": 0.8883745670318604, "rewards/total_composite/std": 0.13054178655147552, "reward": 0.8883745670318604, "reward_std": 0.13054178655147552, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0641026571393013, "sampling/sampling_logp_difference/max": 2.719486713409424, "sampling/importance_sampling_ratio/min": 0.06590858101844788, "sampling/importance_sampling_ratio/mean": 1.0092458724975586, "sampling/importance_sampling_ratio/max": 1.7064915895462036, "entropy": 0.49944454059004784, "clip_ratio/low_mean": 0.018327449448406696, "clip_ratio/low_min": 0.018327449448406696, "clip_ratio/high_mean": 0.030346576124429703, "clip_ratio/high_max": 0.030346576124429703, "clip_ratio/region_mean": 0.0486740255728364, "reward_total_mean": 0.8883745670318604, "reward_meter_mean": 0.9966393113136292, "reward_meter_std": 0.0009264187538065016, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9503968358039856, "reward_repeat_penalty_std": 0.06915634870529175, "reward_total_composite_mean": 0.8883745670318604, "reward_total_composite_std": 0.13054178655147552} {"timestamp_utc": "2026-04-12T00:28:47Z", "mode": "train", "global_step": 1596, "epoch": 0.0641041089287866, "loss": 0.0073, "grad_norm": 6.151648998260498, "learning_rate": 5.1666666666666675e-06, "num_tokens": 3584948.0, "completions/mean_length": 49.75, "completions/min_length": 49.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.75, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.9367859363555908, "rewards/meter/std": 0.005013443063944578, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9367859363555908, "rewards/total_composite/std": 0.005013443063944578, "reward": 0.9367859363555908, "reward_std": 0.005013437941670418, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024813249707221985, "sampling/sampling_logp_difference/max": 1.2509276866912842, "sampling/importance_sampling_ratio/min": 0.2862391471862793, "sampling/importance_sampling_ratio/mean": 1.0016454458236694, "sampling/importance_sampling_ratio/max": 1.7191351652145386, "entropy": 0.11144222784787416, "clip_ratio/low_mean": 0.010051020188257098, "clip_ratio/low_min": 0.010051020188257098, "clip_ratio/high_mean": 0.02265306143090129, "clip_ratio/high_max": 0.02265306143090129, "clip_ratio/region_mean": 0.03270408161915839, "reward_total_mean": 0.9367859363555908, "reward_meter_mean": 0.9367859363555908, "reward_meter_std": 0.005013443063944578, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9367859363555908, "reward_total_composite_std": 0.005013443063944578} {"timestamp_utc": "2026-04-12T00:28:52Z", "mode": "train", "global_step": 1597, "epoch": 0.06414427441057155, "loss": 0.0102, "grad_norm": 4.290965557098389, "learning_rate": 5.163636363636364e-06, "num_tokens": 3587243.0, "completions/mean_length": 111.875, "completions/min_length": 106.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 111.875, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.9952096939086914, "rewards/meter/std": 0.004024881403893232, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.925000011920929, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.9206209182739258, "rewards/total_composite/std": 0.1035769134759903, "reward": 0.9206209182739258, "reward_std": 0.10357693582773209, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.052159715443849564, "sampling/sampling_logp_difference/max": 1.9255342483520508, "sampling/importance_sampling_ratio/min": 0.14579784870147705, "sampling/importance_sampling_ratio/mean": 1.011698842048645, "sampling/importance_sampling_ratio/max": 1.8452407121658325, "entropy": 0.40937453880906105, "clip_ratio/low_mean": 0.011171216145157814, "clip_ratio/low_min": 0.011171216145157814, "clip_ratio/high_mean": 0.02771200449205935, "clip_ratio/high_max": 0.02771200449205935, "clip_ratio/region_mean": 0.038883220637217164, "reward_total_mean": 0.9206209182739258, "reward_meter_mean": 0.9952096939086914, "reward_meter_std": 0.004024881403893232, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.925000011920929, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.9206209182739258, "reward_total_composite_std": 0.1035769134759903} {"timestamp_utc": "2026-04-12T00:28:57Z", "mode": "train", "global_step": 1598, "epoch": 0.0641844398923565, "loss": 0.003, "grad_norm": 2.746960401535034, "learning_rate": 5.160606060606061e-06, "num_tokens": 3589155.0, "completions/mean_length": 64.0, "completions/min_length": 63.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9953999519348145, "rewards/meter/std": 0.0069043696857988834, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9953999519348145, "rewards/total_composite/std": 0.0069043696857988834, "reward": 0.9953999519348145, "reward_std": 0.006904360372573137, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021562010049819946, "sampling/sampling_logp_difference/max": 0.8483424186706543, "sampling/importance_sampling_ratio/min": 0.42812401056289673, "sampling/importance_sampling_ratio/mean": 0.9994773268699646, "sampling/importance_sampling_ratio/max": 1.8938157558441162, "entropy": 0.0873101712204516, "clip_ratio/low_mean": 0.001953125, "clip_ratio/low_min": 0.001953125, "clip_ratio/high_mean": 0.01751898298971355, "clip_ratio/high_max": 0.01751898298971355, "clip_ratio/region_mean": 0.01947210798971355, "reward_total_mean": 0.9953999519348145, "reward_meter_mean": 0.9953999519348145, "reward_meter_std": 0.0069043696857988834, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9953999519348145, "reward_total_composite_std": 0.0069043696857988834} {"timestamp_utc": "2026-04-12T00:29:01Z", "mode": "train", "global_step": 1599, "epoch": 0.06422460537414146, "loss": 0.0135, "grad_norm": 6.9674506187438965, "learning_rate": 5.1575757575757575e-06, "num_tokens": 3590515.0, "completions/mean_length": 37.0, "completions/min_length": 36.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.0, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.9930006265640259, "rewards/meter/std": 0.00047757019638083875, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9930006265640259, "rewards/total_composite/std": 0.00047757019638083875, "reward": 0.9930006265640259, "reward_std": 0.0004775776178576052, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030742382630705833, "sampling/sampling_logp_difference/max": 0.5062961578369141, "sampling/importance_sampling_ratio/min": 0.6027238368988037, "sampling/importance_sampling_ratio/mean": 1.0091636180877686, "sampling/importance_sampling_ratio/max": 1.5318171977996826, "entropy": 0.307000532746315, "clip_ratio/low_mean": 0.006756756920367479, "clip_ratio/low_min": 0.006756756920367479, "clip_ratio/high_mean": 0.017178362933918834, "clip_ratio/high_max": 0.017178362933918834, "clip_ratio/region_mean": 0.023935119854286313, "reward_total_mean": 0.9930006265640259, "reward_meter_mean": 0.9930006265640259, "reward_meter_std": 0.00047757019638083875, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9930006265640259, "reward_total_composite_std": 0.00047757019638083875} {"timestamp_utc": "2026-04-12T00:29:06Z", "mode": "train", "global_step": 1600, "epoch": 0.06426477085592641, "loss": 0.0215, "grad_norm": 6.287996292114258, "learning_rate": 5.154545454545456e-06, "num_tokens": 3592322.0, "completions/mean_length": 64.875, "completions/min_length": 64.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.875, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9355560541152954, "rewards/meter/std": 0.06913071125745773, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9355560541152954, "rewards/total_composite/std": 0.06913071125745773, "reward": 0.9355560541152954, "reward_std": 0.06913069635629654, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03889580816030502, "sampling/sampling_logp_difference/max": 0.9479434490203857, "sampling/importance_sampling_ratio/min": 0.3875372111797333, "sampling/importance_sampling_ratio/mean": 1.003296136856079, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.24194820784032345, "clip_ratio/low_mean": 0.007490954361855984, "clip_ratio/low_min": 0.007490954361855984, "clip_ratio/high_mean": 0.01740056835114956, "clip_ratio/high_max": 0.01740056835114956, "clip_ratio/region_mean": 0.024891522713005543, "reward_total_mean": 0.9355560541152954, "reward_meter_mean": 0.9355560541152954, "reward_meter_std": 0.06913071125745773, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9355560541152954, "reward_total_composite_std": 0.06913071125745773} {"timestamp_utc": "2026-04-12T00:30:12Z", "mode": "eval", "global_step": 1600, "epoch": 0.06426477085592641, "eval_loss": NaN, "eval_runtime": 66.2489, "eval_samples_per_second": 1.57, "eval_steps_per_second": 0.196, "eval_num_tokens": 3592322.0, "eval_completions/mean_length": 192.58653846153845, "eval_completions/min_length": 61.76923076923077, "eval_completions/max_length": 347.3076923076923, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 192.58653846153845, "eval_completions/min_terminated_length": 61.76923076923077, "eval_completions/max_terminated_length": 347.3076923076923, "eval_rewards/meter/mean": 0.7114841112723718, "eval_rewards/meter/std": 0.38725116863273656, "eval_rewards/count_adherence/mean": 0.9018935148532574, "eval_rewards/count_adherence/std": 0.12116303610113952, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/repeat_penalty/mean": 0.8302872364337628, "eval_rewards/repeat_penalty/std": 0.14936749923687714, "eval_rewards/total_composite/mean": 0.5311165887575883, "eval_rewards/total_composite/std": 0.34226179122924805, "eval_reward": 0.5311165887575883, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.0201402991436995, "eval_sampling/sampling_logp_difference/max": 1.060350748208853, "eval_sampling/importance_sampling_ratio/min": 0.35776989047343916, "eval_sampling/importance_sampling_ratio/mean": 1.004677598293011, "eval_sampling/importance_sampling_ratio/max": 1.4436852565178504, "eval_entropy": 0.19453936872574, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5311165887575883, "eval_reward_meter_mean": 0.7114841112723718, "eval_reward_meter_std": 0.38725116863273656, "eval_reward_count_adherence_mean": 0.9018935148532574, "eval_reward_count_adherence_std": 0.12116303610113952, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_repeat_penalty_mean": 0.8302872364337628, "eval_reward_repeat_penalty_std": 0.14936749923687714, "eval_reward_total_composite_mean": 0.5311165887575883, "eval_reward_total_composite_std": 0.34226179122924805} {"timestamp_utc": "2026-04-12T00:30:21Z", "mode": "train", "global_step": 1601, "epoch": 0.06430493633771137, "loss": -0.0132, "grad_norm": 5.907294273376465, "learning_rate": 5.151515151515152e-06, "num_tokens": 3594188.0, "completions/mean_length": 77.25, "completions/min_length": 73.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.25, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9925735592842102, "rewards/meter/std": 0.013519754633307457, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9925735592842102, "rewards/total_composite/std": 0.013519754633307457, "reward": 0.9925735592842102, "reward_std": 0.013519756495952606, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.061646249145269394, "sampling/sampling_logp_difference/max": 3.263657808303833, "sampling/importance_sampling_ratio/min": 0.21933509409427643, "sampling/importance_sampling_ratio/mean": 1.0113800764083862, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4462718777358532, "clip_ratio/low_mean": 0.0034246575087308884, "clip_ratio/low_min": 0.0034246575087308884, "clip_ratio/high_mean": 0.03826867160387337, "clip_ratio/high_max": 0.03826867160387337, "clip_ratio/region_mean": 0.04169332911260426, "reward_total_mean": 0.9925735592842102, "reward_meter_mean": 0.9925735592842102, "reward_meter_std": 0.013519754633307457, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9925735592842102, "reward_total_composite_std": 0.013519754633307457} {"timestamp_utc": "2026-04-12T00:30:27Z", "mode": "train", "global_step": 1602, "epoch": 0.06434510181949632, "loss": -0.0111, "grad_norm": 3.9016387462615967, "learning_rate": 5.148484848484849e-06, "num_tokens": 3596572.0, "completions/mean_length": 133.0, "completions/min_length": 125.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 133.0, "completions/min_terminated_length": 125.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.10229554772377014, "rewards/meter/std": 0.20073863863945007, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.08772411197423935, "rewards/total_composite/std": 0.17203770577907562, "reward": 0.08772411197423935, "reward_std": 0.17203769087791443, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06598487496376038, "sampling/sampling_logp_difference/max": 1.3837254047393799, "sampling/importance_sampling_ratio/min": 0.27479445934295654, "sampling/importance_sampling_ratio/mean": 1.0097557306289673, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4912920892238617, "clip_ratio/low_mean": 0.038138989359140396, "clip_ratio/low_min": 0.038138989359140396, "clip_ratio/high_mean": 0.016364703187718987, "clip_ratio/high_max": 0.016364703187718987, "clip_ratio/region_mean": 0.054503692546859384, "reward_total_mean": 0.08772411197423935, "reward_meter_mean": 0.10229554772377014, "reward_meter_std": 0.20073863863945007, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.08772411197423935, "reward_total_composite_std": 0.17203770577907562} {"timestamp_utc": "2026-04-12T00:30:33Z", "mode": "train", "global_step": 1603, "epoch": 0.06438526730128127, "loss": -0.0097, "grad_norm": 2.740659713745117, "learning_rate": 5.145454545454546e-06, "num_tokens": 3598835.0, "completions/mean_length": 104.875, "completions/min_length": 102.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 104.875, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.9987785816192627, "rewards/meter/std": 0.00020610357751138508, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8250000476837158, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8239848613739014, "rewards/total_composite/std": 0.07050739973783493, "reward": 0.8239848613739014, "reward_std": 0.07050740718841553, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019803691655397415, "sampling/sampling_logp_difference/max": 0.6796939373016357, "sampling/importance_sampling_ratio/min": 0.5668526887893677, "sampling/importance_sampling_ratio/mean": 1.0108842849731445, "sampling/importance_sampling_ratio/max": 1.973273754119873, "entropy": 0.10506064165383577, "clip_ratio/low_mean": 0.01308055012486875, "clip_ratio/low_min": 0.01308055012486875, "clip_ratio/high_mean": 0.002314814832061529, "clip_ratio/high_max": 0.002314814832061529, "clip_ratio/region_mean": 0.01539536495693028, "reward_total_mean": 0.8239848613739014, "reward_meter_mean": 0.9987785816192627, "reward_meter_std": 0.00020610357751138508, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8250000476837158, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.8239848613739014, "reward_total_composite_std": 0.07050739973783493} {"timestamp_utc": "2026-04-12T00:30:39Z", "mode": "train", "global_step": 1604, "epoch": 0.06442543278306623, "loss": -0.0024, "grad_norm": 1.9800543785095215, "learning_rate": 5.142424242424243e-06, "num_tokens": 3601733.0, "completions/mean_length": 178.25, "completions/min_length": 172.0, "completions/max_length": 180.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 178.25, "completions/min_terminated_length": 172.0, "completions/max_terminated_length": 180.0, "rewards/meter/mean": 0.9987528324127197, "rewards/meter/std": 0.00028449000092223287, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7361111044883728, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.7351874709129333, "rewards/total_composite/std": 0.0573553703725338, "reward": 0.7351874709129333, "reward_std": 0.0573553666472435, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01709139533340931, "sampling/sampling_logp_difference/max": 1.673609972000122, "sampling/importance_sampling_ratio/min": 0.18756872415542603, "sampling/importance_sampling_ratio/mean": 1.0028082132339478, "sampling/importance_sampling_ratio/max": 1.682837963104248, "entropy": 0.08005253970623016, "clip_ratio/low_mean": 0.00780038780067116, "clip_ratio/low_min": 0.00780038780067116, "clip_ratio/high_mean": 0.006316610379144549, "clip_ratio/high_max": 0.006316610379144549, "clip_ratio/region_mean": 0.01411699817981571, "reward_total_mean": 0.7351874709129333, "reward_meter_mean": 0.9987528324127197, "reward_meter_std": 0.00028449000092223287, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7361111044883728, "reward_repeat_penalty_std": 0.05750546231865883, "reward_total_composite_mean": 0.7351874709129333, "reward_total_composite_std": 0.0573553703725338} {"timestamp_utc": "2026-04-12T00:30:44Z", "mode": "train", "global_step": 1605, "epoch": 0.06446559826485118, "loss": 0.0008, "grad_norm": 0.7806724905967712, "learning_rate": 5.139393939393939e-06, "num_tokens": 3603520.0, "completions/mean_length": 71.375, "completions/min_length": 70.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.375, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9987327456474304, "rewards/meter/std": 6.793010834371671e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9987327456474304, "rewards/total_composite/std": 6.793010834371671e-05, "reward": 0.9987327456474304, "reward_std": 6.793974171159789e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015499016270041466, "sampling/sampling_logp_difference/max": 0.9751527309417725, "sampling/importance_sampling_ratio/min": 0.3771347403526306, "sampling/importance_sampling_ratio/mean": 1.0033406019210815, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07925032824277878, "clip_ratio/low_mean": 0.0035714285913854837, "clip_ratio/low_min": 0.0035714285913854837, "clip_ratio/high_mean": 0.008729460067115724, "clip_ratio/high_max": 0.008729460067115724, "clip_ratio/region_mean": 0.012300888658501208, "reward_total_mean": 0.9987327456474304, "reward_meter_mean": 0.9987327456474304, "reward_meter_std": 6.793010834371671e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9987327456474304, "reward_total_composite_std": 6.793010834371671e-05} {"timestamp_utc": "2026-04-12T00:30:50Z", "mode": "train", "global_step": 1606, "epoch": 0.06450576374663614, "loss": -0.0107, "grad_norm": 1.1906522512435913, "learning_rate": 5.1363636363636375e-06, "num_tokens": 3605709.0, "completions/mean_length": 99.625, "completions/min_length": 99.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.625, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.9968845248222351, "rewards/meter/std": 0.0012903602328151464, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8250000476837158, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8223729729652405, "rewards/total_composite/std": 0.06959936022758484, "reward": 0.8223729729652405, "reward_std": 0.06959936767816544, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006160242948681116, "sampling/sampling_logp_difference/max": 0.7100238800048828, "sampling/importance_sampling_ratio/min": 0.49163246154785156, "sampling/importance_sampling_ratio/mean": 1.002079963684082, "sampling/importance_sampling_ratio/max": 1.2040398120880127, "entropy": 0.050112520810216665, "clip_ratio/low_mean": 0.0012499999720603228, "clip_ratio/low_min": 0.0012499999720603228, "clip_ratio/high_mean": 0.0012135922443121672, "clip_ratio/high_max": 0.0012135922443121672, "clip_ratio/region_mean": 0.00246359221637249, "reward_total_mean": 0.8223729729652405, "reward_meter_mean": 0.9968845248222351, "reward_meter_std": 0.0012903602328151464, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8250000476837158, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.8223729729652405, "reward_total_composite_std": 0.06959936022758484} {"timestamp_utc": "2026-04-12T00:30:56Z", "mode": "train", "global_step": 1607, "epoch": 0.06454592922842109, "loss": 0.0229, "grad_norm": 4.926582336425781, "learning_rate": 5.133333333333334e-06, "num_tokens": 3608643.0, "completions/mean_length": 171.75, "completions/min_length": 163.0, "completions/max_length": 183.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 171.75, "completions/min_terminated_length": 163.0, "completions/max_terminated_length": 183.0, "rewards/meter/mean": 0.937900185585022, "rewards/meter/std": 0.15882940590381622, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9319444298744202, "rewards/repeat_penalty/std": 0.08195958286523819, "rewards/total_composite/mean": 0.8702612519264221, "rewards/total_composite/std": 0.15232259035110474, "reward": 0.8702612519264221, "reward_std": 0.15232259035110474, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.058821044862270355, "sampling/sampling_logp_difference/max": 1.4120397567749023, "sampling/importance_sampling_ratio/min": 0.2436458021402359, "sampling/importance_sampling_ratio/mean": 1.0150550603866577, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.48374373838305473, "clip_ratio/low_mean": 0.010939412750303745, "clip_ratio/low_min": 0.010939412750303745, "clip_ratio/high_mean": 0.035856032744050026, "clip_ratio/high_max": 0.035856032744050026, "clip_ratio/region_mean": 0.04679544549435377, "reward_total_mean": 0.8702612519264221, "reward_meter_mean": 0.937900185585022, "reward_meter_std": 0.15882940590381622, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9319444298744202, "reward_repeat_penalty_std": 0.08195958286523819, "reward_total_composite_mean": 0.8702612519264221, "reward_total_composite_std": 0.15232259035110474} {"timestamp_utc": "2026-04-12T00:31:02Z", "mode": "train", "global_step": 1608, "epoch": 0.06458609471020604, "loss": -0.02, "grad_norm": 2.1128990650177, "learning_rate": 5.130303030303031e-06, "num_tokens": 3611925.0, "completions/mean_length": 217.25, "completions/min_length": 207.0, "completions/max_length": 239.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 217.25, "completions/min_terminated_length": 207.0, "completions/max_terminated_length": 239.0, "rewards/meter/mean": 0.9952753782272339, "rewards/meter/std": 0.0022598514333367348, "rewards/count_adherence/mean": 0.9642857313156128, "rewards/count_adherence/std": 0.06613000482320786, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8250916004180908, "rewards/repeat_penalty/std": 0.05420750379562378, "rewards/total_composite/mean": 0.7915375232696533, "rewards/total_composite/std": 0.07283394038677216, "reward": 0.7915375232696533, "reward_std": 0.07283391803503036, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.036408815532922745, "sampling/sampling_logp_difference/max": 8.568058013916016, "sampling/importance_sampling_ratio/min": 0.00019008143863175064, "sampling/importance_sampling_ratio/mean": 0.999096691608429, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08922967594116926, "clip_ratio/low_mean": 0.005352868989575654, "clip_ratio/low_min": 0.005352868989575654, "clip_ratio/high_mean": 0.014548269449733198, "clip_ratio/high_max": 0.014548269449733198, "clip_ratio/region_mean": 0.019901138439308852, "reward_total_mean": 0.7915375232696533, "reward_meter_mean": 0.9952753782272339, "reward_meter_std": 0.0022598514333367348, "reward_count_adherence_mean": 0.9642857313156128, "reward_count_adherence_std": 0.06613000482320786, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8250916004180908, "reward_repeat_penalty_std": 0.05420750379562378, "reward_total_composite_mean": 0.7915375232696533, "reward_total_composite_std": 0.07283394038677216} {"timestamp_utc": "2026-04-12T00:31:07Z", "mode": "train", "global_step": 1609, "epoch": 0.064626260191991, "loss": 0.002, "grad_norm": 3.462772846221924, "learning_rate": 5.1272727272727275e-06, "num_tokens": 3613586.0, "completions/mean_length": 60.625, "completions/min_length": 59.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.625, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9920006990432739, "rewards/meter/std": 0.00801026076078415, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9506175518035889, "rewards/total_composite/std": 0.1168404370546341, "reward": 0.9506175518035889, "reward_std": 0.11684045940637589, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02973771281540394, "sampling/sampling_logp_difference/max": 0.9864425659179688, "sampling/importance_sampling_ratio/min": 0.37290093302726746, "sampling/importance_sampling_ratio/mean": 1.006797194480896, "sampling/importance_sampling_ratio/max": 1.5062812566757202, "entropy": 0.17171061784029007, "clip_ratio/low_mean": 0.004098360426723957, "clip_ratio/low_min": 0.004098360426723957, "clip_ratio/high_mean": 0.016640397254377604, "clip_ratio/high_max": 0.016640397254377604, "clip_ratio/region_mean": 0.02073875768110156, "reward_total_mean": 0.9506175518035889, "reward_meter_mean": 0.9920006990432739, "reward_meter_std": 0.00801026076078415, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.9506175518035889, "reward_total_composite_std": 0.1168404370546341} {"timestamp_utc": "2026-04-12T00:31:11Z", "mode": "train", "global_step": 1610, "epoch": 0.06466642567377595, "loss": 0.0039, "grad_norm": 3.0595595836639404, "learning_rate": 5.124242424242425e-06, "num_tokens": 3615443.0, "completions/mean_length": 66.125, "completions/min_length": 64.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.125, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9358171224594116, "rewards/meter/std": 0.018078165128827095, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9358171224594116, "rewards/total_composite/std": 0.018078165128827095, "reward": 0.9358171224594116, "reward_std": 0.01807817816734314, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03475549817085266, "sampling/sampling_logp_difference/max": 1.2117633819580078, "sampling/importance_sampling_ratio/min": 0.297671914100647, "sampling/importance_sampling_ratio/mean": 1.0040435791015625, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.24806034937500954, "clip_ratio/low_mean": 0.013008192996494472, "clip_ratio/low_min": 0.013008192996494472, "clip_ratio/high_mean": 0.013322061393409967, "clip_ratio/high_max": 0.013322061393409967, "clip_ratio/region_mean": 0.02633025438990444, "reward_total_mean": 0.9358171224594116, "reward_meter_mean": 0.9358171224594116, "reward_meter_std": 0.018078165128827095, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9358171224594116, "reward_total_composite_std": 0.018078165128827095} {"timestamp_utc": "2026-04-12T00:31:17Z", "mode": "train", "global_step": 1611, "epoch": 0.0647065911555609, "loss": 0.0158, "grad_norm": 4.593719959259033, "learning_rate": 5.121212121212121e-06, "num_tokens": 3617986.0, "completions/mean_length": 141.875, "completions/min_length": 133.0, "completions/max_length": 155.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 141.875, "completions/min_terminated_length": 133.0, "completions/max_terminated_length": 155.0, "rewards/meter/mean": 0.9247380495071411, "rewards/meter/std": 0.13655264675617218, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.86052405834198, "rewards/total_composite/std": 0.14909514784812927, "reward": 0.86052405834198, "reward_std": 0.14909514784812927, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07056324183940887, "sampling/sampling_logp_difference/max": 7.457266807556152, "sampling/importance_sampling_ratio/min": 0.0005772316944785416, "sampling/importance_sampling_ratio/mean": 1.0124247074127197, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5591295659542084, "clip_ratio/low_mean": 0.014760382240638137, "clip_ratio/low_min": 0.014760382240638137, "clip_ratio/high_mean": 0.016017435351386666, "clip_ratio/high_max": 0.016017435351386666, "clip_ratio/region_mean": 0.030777817592024803, "reward_total_mean": 0.86052405834198, "reward_meter_mean": 0.9247380495071411, "reward_meter_std": 0.13655264675617218, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.86052405834198, "reward_total_composite_std": 0.14909514784812927} {"timestamp_utc": "2026-04-12T00:31:21Z", "mode": "train", "global_step": 1612, "epoch": 0.06474675663734586, "loss": -0.0037, "grad_norm": 1.593169927597046, "learning_rate": 5.1181818181818185e-06, "num_tokens": 3619751.0, "completions/mean_length": 57.625, "completions/min_length": 55.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.625, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9934520721435547, "rewards/meter/std": 3.5445533285383135e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9934520721435547, "rewards/total_composite/std": 3.5445533285383135e-05, "reward": 0.9934520721435547, "reward_std": 3.5451525036478415e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003705526702105999, "sampling/sampling_logp_difference/max": 0.46508365869522095, "sampling/importance_sampling_ratio/min": 0.6644558906555176, "sampling/importance_sampling_ratio/mean": 1.0014033317565918, "sampling/importance_sampling_ratio/max": 1.5921474695205688, "entropy": 0.018967063282616436, "clip_ratio/low_mean": 0.0022727272007614374, "clip_ratio/low_min": 0.0022727272007614374, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0022727272007614374, "reward_total_mean": 0.9934520721435547, "reward_meter_mean": 0.9934520721435547, "reward_meter_std": 3.5445533285383135e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9934520721435547, "reward_total_composite_std": 3.5445533285383135e-05} {"timestamp_utc": "2026-04-12T00:31:27Z", "mode": "train", "global_step": 1613, "epoch": 0.06478692211913081, "loss": -0.0043, "grad_norm": 7.702324390411377, "learning_rate": 5.115151515151515e-06, "num_tokens": 3621517.0, "completions/mean_length": 63.75, "completions/min_length": 63.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.75, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9974002838134766, "rewards/meter/std": 0.0021041962318122387, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9974002838134766, "rewards/total_composite/std": 0.0021041962318122387, "reward": 0.9974002838134766, "reward_std": 0.002104192739352584, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01148869190365076, "sampling/sampling_logp_difference/max": 0.7638870477676392, "sampling/importance_sampling_ratio/min": 0.46585214138031006, "sampling/importance_sampling_ratio/mean": 1.0017887353897095, "sampling/importance_sampling_ratio/max": 1.317487359046936, "entropy": 0.056046845857053995, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.001953125, "clip_ratio/high_max": 0.001953125, "clip_ratio/region_mean": 0.001953125, "reward_total_mean": 0.9974002838134766, "reward_meter_mean": 0.9974002838134766, "reward_meter_std": 0.0021041962318122387, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9974002838134766, "reward_total_composite_std": 0.0021041962318122387} {"timestamp_utc": "2026-04-12T00:31:33Z", "mode": "train", "global_step": 1614, "epoch": 0.06482708760091577, "loss": 0.0583, "grad_norm": 3.499596118927002, "learning_rate": 5.112121212121213e-06, "num_tokens": 3624361.0, "completions/mean_length": 169.5, "completions/min_length": 154.0, "completions/max_length": 198.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 169.5, "completions/min_terminated_length": 154.0, "completions/max_terminated_length": 198.0, "rewards/meter/mean": 0.9390597343444824, "rewards/meter/std": 0.12540623545646667, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7876983880996704, "rewards/repeat_penalty/std": 0.13440276682376862, "rewards/total_composite/mean": 0.6842913627624512, "rewards/total_composite/std": 0.13273297250270844, "reward": 0.6842913627624512, "reward_std": 0.13273297250270844, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05148252472281456, "sampling/sampling_logp_difference/max": 3.8801069259643555, "sampling/importance_sampling_ratio/min": 0.020648617297410965, "sampling/importance_sampling_ratio/mean": 1.012567162513733, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3938692733645439, "clip_ratio/low_mean": 0.018685899674892426, "clip_ratio/low_min": 0.018685899674892426, "clip_ratio/high_mean": 0.02137809677515179, "clip_ratio/high_max": 0.02137809677515179, "clip_ratio/region_mean": 0.040063996450044215, "reward_total_mean": 0.6842913627624512, "reward_meter_mean": 0.9390597343444824, "reward_meter_std": 0.12540623545646667, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7876983880996704, "reward_repeat_penalty_std": 0.13440276682376862, "reward_total_composite_mean": 0.6842913627624512, "reward_total_composite_std": 0.13273297250270844} {"timestamp_utc": "2026-04-12T00:31:37Z", "mode": "train", "global_step": 1615, "epoch": 0.06486725308270072, "loss": 0.0083, "grad_norm": 9.031867027282715, "learning_rate": 5.109090909090909e-06, "num_tokens": 3626009.0, "completions/mean_length": 58.0, "completions/min_length": 58.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.992936372756958, "rewards/meter/std": 0.0014942112611606717, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.992936372756958, "rewards/total_composite/std": 0.0014942112611606717, "reward": 0.992936372756958, "reward_std": 0.0014942261623218656, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008410189300775528, "sampling/sampling_logp_difference/max": 0.7060290575027466, "sampling/importance_sampling_ratio/min": 0.5572132468223572, "sampling/importance_sampling_ratio/mean": 1.0016261339187622, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.02835727483034134, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.006465517217293382, "clip_ratio/high_max": 0.006465517217293382, "clip_ratio/region_mean": 0.006465517217293382, "reward_total_mean": 0.992936372756958, "reward_meter_mean": 0.992936372756958, "reward_meter_std": 0.0014942112611606717, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.992936372756958, "reward_total_composite_std": 0.0014942112611606717} {"timestamp_utc": "2026-04-12T00:31:42Z", "mode": "train", "global_step": 1616, "epoch": 0.06490741856448567, "loss": 0.02, "grad_norm": 3.926830530166626, "learning_rate": 5.106060606060607e-06, "num_tokens": 3627780.0, "completions/mean_length": 71.375, "completions/min_length": 68.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.375, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9946064352989197, "rewards/meter/std": 0.0011085572186857462, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9946064352989197, "rewards/total_composite/std": 0.0011085572186857462, "reward": 0.9946064352989197, "reward_std": 0.0011085484875366092, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009593191556632519, "sampling/sampling_logp_difference/max": 0.8193645477294922, "sampling/importance_sampling_ratio/min": 0.4407116174697876, "sampling/importance_sampling_ratio/mean": 1.0016788244247437, "sampling/importance_sampling_ratio/max": 1.385042428970337, "entropy": 0.07067765574902296, "clip_ratio/low_mean": 0.0017361111240461469, "clip_ratio/low_min": 0.0017361111240461469, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0017361111240461469, "reward_total_mean": 0.9946064352989197, "reward_meter_mean": 0.9946064352989197, "reward_meter_std": 0.0011085572186857462, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9946064352989197, "reward_total_composite_std": 0.0011085572186857462} {"timestamp_utc": "2026-04-12T00:31:53Z", "mode": "train", "global_step": 1617, "epoch": 0.06494758404627063, "loss": -0.1647, "grad_norm": 1.3416348695755005, "learning_rate": 5.103030303030303e-06, "num_tokens": 3633068.0, "completions/mean_length": 445.0, "completions/min_length": 421.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 435.4285888671875, "completions/min_terminated_length": 421.0, "completions/max_terminated_length": 459.0, "rewards/meter/mean": 0.9931340217590332, "rewards/meter/std": 0.005852298345416784, "rewards/count_adherence/mean": 0.671875, "rewards/count_adherence/std": 0.07281029969453812, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 0.7910888195037842, "rewards/repeat_penalty/std": 0.1950831413269043, "rewards/total_composite/mean": 0.4080624282360077, "rewards/total_composite/std": 0.282764196395874, "reward": 0.4080624282360077, "reward_std": 0.282764196395874, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04456779360771179, "sampling/sampling_logp_difference/max": 3.5916011333465576, "sampling/importance_sampling_ratio/min": 0.02755417674779892, "sampling/importance_sampling_ratio/mean": 1.0104022026062012, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.30827769078314304, "clip_ratio/low_mean": 0.003343421034514904, "clip_ratio/low_min": 0.003343421034514904, "clip_ratio/high_mean": 0.023898042272776365, "clip_ratio/high_max": 0.023898042272776365, "clip_ratio/region_mean": 0.02724146330729127, "reward_total_mean": 0.4080624282360077, "reward_meter_mean": 0.9931340217590332, "reward_meter_std": 0.005852298345416784, "reward_count_adherence_mean": 0.671875, "reward_count_adherence_std": 0.07281029969453812, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 0.7910888195037842, "reward_repeat_penalty_std": 0.1950831413269043, "reward_total_composite_mean": 0.4080624282360077, "reward_total_composite_std": 0.282764196395874} {"timestamp_utc": "2026-04-12T00:32:02Z", "mode": "train", "global_step": 1618, "epoch": 0.06498774952805558, "loss": -0.0118, "grad_norm": 1.6642099618911743, "learning_rate": 5.1e-06, "num_tokens": 3637584.0, "completions/mean_length": 389.5, "completions/min_length": 376.0, "completions/max_length": 402.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 389.5, "completions/min_terminated_length": 376.0, "completions/max_terminated_length": 402.0, "rewards/meter/mean": 0.9942715167999268, "rewards/meter/std": 0.005808908957988024, "rewards/count_adherence/mean": 0.7142857313156128, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8421052694320679, "rewards/repeat_penalty/std": 0.14886459708213806, "rewards/total_composite/mean": 0.597885012626648, "rewards/total_composite/std": 0.10485604405403137, "reward": 0.597885012626648, "reward_std": 0.10485603660345078, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.044363927096128464, "sampling/sampling_logp_difference/max": 2.440901279449463, "sampling/importance_sampling_ratio/min": 0.08708232641220093, "sampling/importance_sampling_ratio/mean": 1.0039002895355225, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3395629785954952, "clip_ratio/low_mean": 0.009485365822911263, "clip_ratio/low_min": 0.009485365822911263, "clip_ratio/high_mean": 0.02332985820248723, "clip_ratio/high_max": 0.02332985820248723, "clip_ratio/region_mean": 0.03281522402539849, "reward_total_mean": 0.597885012626648, "reward_meter_mean": 0.9942715167999268, "reward_meter_std": 0.005808908957988024, "reward_count_adherence_mean": 0.7142857313156128, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8421052694320679, "reward_repeat_penalty_std": 0.14886459708213806, "reward_total_composite_mean": 0.597885012626648, "reward_total_composite_std": 0.10485604405403137} {"timestamp_utc": "2026-04-12T00:32:07Z", "mode": "train", "global_step": 1619, "epoch": 0.06502791500984054, "loss": 0.0001, "grad_norm": 10.757038116455078, "learning_rate": 5.096969696969697e-06, "num_tokens": 3639232.0, "completions/mean_length": 59.0, "completions/min_length": 57.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.0, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9416102766990662, "rewards/meter/std": 0.15220597386360168, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9416102766990662, "rewards/total_composite/std": 0.15220597386360168, "reward": 0.9416102766990662, "reward_std": 0.1522059589624405, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018271297216415405, "sampling/sampling_logp_difference/max": 0.5301923751831055, "sampling/importance_sampling_ratio/min": 0.5884917974472046, "sampling/importance_sampling_ratio/mean": 1.007680892944336, "sampling/importance_sampling_ratio/max": 1.6727921962738037, "entropy": 0.12759912386536598, "clip_ratio/low_mean": 0.0021186440717428923, "clip_ratio/low_min": 0.0021186440717428923, "clip_ratio/high_mean": 0.004385964944958687, "clip_ratio/high_max": 0.004385964944958687, "clip_ratio/region_mean": 0.006504609016701579, "reward_total_mean": 0.9416102766990662, "reward_meter_mean": 0.9416102766990662, "reward_meter_std": 0.15220597386360168, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9416102766990662, "reward_total_composite_std": 0.15220597386360168} {"timestamp_utc": "2026-04-12T00:32:12Z", "mode": "train", "global_step": 1620, "epoch": 0.06506808049162549, "loss": 0.032, "grad_norm": 4.442444801330566, "learning_rate": 5.093939393939395e-06, "num_tokens": 3641223.0, "completions/mean_length": 77.875, "completions/min_length": 75.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.875, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9966658353805542, "rewards/meter/std": 0.0016763400053605437, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9966658353805542, "rewards/total_composite/std": 0.0016763400053605437, "reward": 0.9966658353805542, "reward_std": 0.0016763500170782208, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06199686974287033, "sampling/sampling_logp_difference/max": 1.449620246887207, "sampling/importance_sampling_ratio/min": 0.23465938866138458, "sampling/importance_sampling_ratio/mean": 1.0116773843765259, "sampling/importance_sampling_ratio/max": 1.952990174293518, "entropy": 0.5332462079823017, "clip_ratio/low_mean": 0.01416793093085289, "clip_ratio/low_min": 0.01416793093085289, "clip_ratio/high_mean": 0.02794372313655913, "clip_ratio/high_max": 0.02794372313655913, "clip_ratio/region_mean": 0.04211165406741202, "reward_total_mean": 0.9966658353805542, "reward_meter_mean": 0.9966658353805542, "reward_meter_std": 0.0016763400053605437, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9966658353805542, "reward_total_composite_std": 0.0016763400053605437} {"timestamp_utc": "2026-04-12T00:32:16Z", "mode": "train", "global_step": 1621, "epoch": 0.06510824597341044, "loss": 0.2585, "grad_norm": 9.210960388183594, "learning_rate": 5.090909090909091e-06, "num_tokens": 3642899.0, "completions/mean_length": 46.5, "completions/min_length": 39.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.5, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.959256649017334, "rewards/meter/std": 0.07027505338191986, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.959256649017334, "rewards/total_composite/std": 0.07027505338191986, "reward": 0.959256649017334, "reward_std": 0.07027503848075867, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0790342465043068, "sampling/sampling_logp_difference/max": 1.2574310302734375, "sampling/importance_sampling_ratio/min": 0.2843836545944214, "sampling/importance_sampling_ratio/mean": 1.0092580318450928, "sampling/importance_sampling_ratio/max": 1.8360445499420166, "entropy": 0.7850339710712433, "clip_ratio/low_mean": 0.02026098920032382, "clip_ratio/low_min": 0.02026098920032382, "clip_ratio/high_mean": 0.0513387790415436, "clip_ratio/high_max": 0.0513387790415436, "clip_ratio/region_mean": 0.07159976824186742, "reward_total_mean": 0.959256649017334, "reward_meter_mean": 0.959256649017334, "reward_meter_std": 0.07027505338191986, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.959256649017334, "reward_total_composite_std": 0.07027505338191986} {"timestamp_utc": "2026-04-12T00:32:21Z", "mode": "train", "global_step": 1622, "epoch": 0.0651484114551954, "loss": -0.1515, "grad_norm": 5.405808925628662, "learning_rate": 5.0878787878787885e-06, "num_tokens": 3644592.0, "completions/mean_length": 55.625, "completions/min_length": 31.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.625, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.993726372718811, "rewards/meter/std": 0.0034743899013847113, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9314955472946167, "rewards/total_composite/std": 0.17525316774845123, "reward": 0.9314955472946167, "reward_std": 0.17525316774845123, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02844316139817238, "sampling/sampling_logp_difference/max": 1.0102331638336182, "sampling/importance_sampling_ratio/min": 0.3641340732574463, "sampling/importance_sampling_ratio/mean": 1.0055111646652222, "sampling/importance_sampling_ratio/max": 1.6441330909729004, "entropy": 0.16878793388605118, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.01896783267147839, "clip_ratio/high_max": 0.01896783267147839, "clip_ratio/region_mean": 0.01896783267147839, "reward_total_mean": 0.9314955472946167, "reward_meter_mean": 0.993726372718811, "reward_meter_std": 0.0034743899013847113, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9314955472946167, "reward_total_composite_std": 0.17525316774845123} {"timestamp_utc": "2026-04-12T00:32:26Z", "mode": "train", "global_step": 1623, "epoch": 0.06518857693698035, "loss": 0.0053, "grad_norm": 4.352970123291016, "learning_rate": 5.084848484848486e-06, "num_tokens": 3646304.0, "completions/mean_length": 72.0, "completions/min_length": 69.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.0, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.8835433125495911, "rewards/meter/std": 0.2843259274959564, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8835433125495911, "rewards/total_composite/std": 0.2843259274959564, "reward": 0.8835433125495911, "reward_std": 0.2843259274959564, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.061763398349285126, "sampling/sampling_logp_difference/max": 2.2054567337036133, "sampling/importance_sampling_ratio/min": 0.1102001741528511, "sampling/importance_sampling_ratio/mean": 1.0209115743637085, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6054626666009426, "clip_ratio/low_mean": 0.0051369862630963326, "clip_ratio/low_min": 0.0051369862630963326, "clip_ratio/high_mean": 0.04867990920320153, "clip_ratio/high_max": 0.04867990920320153, "clip_ratio/region_mean": 0.053816895466297865, "reward_total_mean": 0.8835433125495911, "reward_meter_mean": 0.8835433125495911, "reward_meter_std": 0.2843259274959564, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8835433125495911, "reward_total_composite_std": 0.2843259274959564} {"timestamp_utc": "2026-04-12T00:32:35Z", "mode": "train", "global_step": 1624, "epoch": 0.0652287424187653, "loss": -0.0003, "grad_norm": 2.14442777633667, "learning_rate": 5.081818181818182e-06, "num_tokens": 3650768.0, "completions/mean_length": 336.0, "completions/min_length": 306.0, "completions/max_length": 347.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 336.0, "completions/min_terminated_length": 306.0, "completions/max_terminated_length": 347.0, "rewards/meter/mean": 0.9964303374290466, "rewards/meter/std": 0.0028279528487473726, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.046290989965200424, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9401960372924805, "rewards/repeat_penalty/std": 0.04455278813838959, "rewards/total_composite/mean": 0.8190490007400513, "rewards/total_composite/std": 0.04663979262113571, "reward": 0.8190490007400513, "reward_std": 0.04663977771997452, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.058607399463653564, "sampling/sampling_logp_difference/max": 1.395674467086792, "sampling/importance_sampling_ratio/min": 0.24766594171524048, "sampling/importance_sampling_ratio/mean": 1.01429283618927, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5225833132863045, "clip_ratio/low_mean": 0.016402944223955274, "clip_ratio/low_min": 0.016402944223955274, "clip_ratio/high_mean": 0.022054610773921013, "clip_ratio/high_max": 0.022054610773921013, "clip_ratio/region_mean": 0.038457554997876287, "reward_total_mean": 0.8190490007400513, "reward_meter_mean": 0.9964303374290466, "reward_meter_std": 0.0028279528487473726, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.046290989965200424, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9401960372924805, "reward_repeat_penalty_std": 0.04455278813838959, "reward_total_composite_mean": 0.8190490007400513, "reward_total_composite_std": 0.04663979262113571} {"timestamp_utc": "2026-04-12T00:32:41Z", "mode": "train", "global_step": 1625, "epoch": 0.06526890790055026, "loss": -0.0375, "grad_norm": 4.290684223175049, "learning_rate": 5.078787878787879e-06, "num_tokens": 3653643.0, "completions/mean_length": 171.375, "completions/min_length": 142.0, "completions/max_length": 189.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 171.375, "completions/min_terminated_length": 142.0, "completions/max_terminated_length": 189.0, "rewards/meter/mean": 0.9575715065002441, "rewards/meter/std": 0.10626022517681122, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9682539701461792, "rewards/repeat_penalty/std": 0.059391383081674576, "rewards/total_composite/mean": 0.8372185230255127, "rewards/total_composite/std": 0.15995363891124725, "reward": 0.8372185230255127, "reward_std": 0.15995362401008606, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09049484878778458, "sampling/sampling_logp_difference/max": 3.279475450515747, "sampling/importance_sampling_ratio/min": 0.037647999823093414, "sampling/importance_sampling_ratio/mean": 1.009111762046814, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7438356392085552, "clip_ratio/low_mean": 0.02405611891299486, "clip_ratio/low_min": 0.02405611891299486, "clip_ratio/high_mean": 0.04015351925045252, "clip_ratio/high_max": 0.04015351925045252, "clip_ratio/region_mean": 0.06420963816344738, "reward_total_mean": 0.8372185230255127, "reward_meter_mean": 0.9575715065002441, "reward_meter_std": 0.10626022517681122, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9682539701461792, "reward_repeat_penalty_std": 0.059391383081674576, "reward_total_composite_mean": 0.8372185230255127, "reward_total_composite_std": 0.15995363891124725} {"timestamp_utc": "2026-04-12T00:32:46Z", "mode": "train", "global_step": 1626, "epoch": 0.06530907338233521, "loss": 0.0078, "grad_norm": 3.822707414627075, "learning_rate": 5.075757575757576e-06, "num_tokens": 3655935.0, "completions/mean_length": 96.5, "completions/min_length": 92.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.5, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.9368202686309814, "rewards/meter/std": 0.026791444048285484, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9368202686309814, "rewards/total_composite/std": 0.026791444048285484, "reward": 0.9368202686309814, "reward_std": 0.02679145336151123, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05230702832341194, "sampling/sampling_logp_difference/max": 0.9677839279174805, "sampling/importance_sampling_ratio/min": 0.37992405891418457, "sampling/importance_sampling_ratio/mean": 1.0122920274734497, "sampling/importance_sampling_ratio/max": 1.796831727027893, "entropy": 0.45825108513236046, "clip_ratio/low_mean": 0.017135416390374303, "clip_ratio/low_min": 0.017135416390374303, "clip_ratio/high_mean": 0.01032647315878421, "clip_ratio/high_max": 0.01032647315878421, "clip_ratio/region_mean": 0.027461889549158514, "reward_total_mean": 0.9368202686309814, "reward_meter_mean": 0.9368202686309814, "reward_meter_std": 0.026791444048285484, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9368202686309814, "reward_total_composite_std": 0.026791444048285484} {"timestamp_utc": "2026-04-12T00:32:51Z", "mode": "train", "global_step": 1627, "epoch": 0.06534923886412017, "loss": 0.0042, "grad_norm": 6.298683166503906, "learning_rate": 5.072727272727274e-06, "num_tokens": 3657666.0, "completions/mean_length": 59.375, "completions/min_length": 57.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.375, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.8852929472923279, "rewards/meter/std": 0.30933257937431335, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8852929472923279, "rewards/total_composite/std": 0.30933257937431335, "reward": 0.8852929472923279, "reward_std": 0.30933254957199097, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025638654828071594, "sampling/sampling_logp_difference/max": 0.9846286773681641, "sampling/importance_sampling_ratio/min": 0.37357792258262634, "sampling/importance_sampling_ratio/mean": 1.0004394054412842, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1392963044345379, "clip_ratio/low_mean": 0.004098360426723957, "clip_ratio/low_min": 0.004098360426723957, "clip_ratio/high_mean": 0.014909721445292234, "clip_ratio/high_max": 0.014909721445292234, "clip_ratio/region_mean": 0.01900808187201619, "reward_total_mean": 0.8852929472923279, "reward_meter_mean": 0.8852929472923279, "reward_meter_std": 0.30933257937431335, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8852929472923279, "reward_total_composite_std": 0.30933257937431335} {"timestamp_utc": "2026-04-12T00:32:55Z", "mode": "train", "global_step": 1628, "epoch": 0.06538940434590514, "loss": 0.0099, "grad_norm": 3.436232328414917, "learning_rate": 5.06969696969697e-06, "num_tokens": 3659244.0, "completions/mean_length": 35.25, "completions/min_length": 35.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.25, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9971714019775391, "rewards/meter/std": 0.0016039953334257007, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9971714019775391, "rewards/total_composite/std": 0.0016039953334257007, "reward": 0.9971714019775391, "reward_std": 0.0016039892798289657, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0144999660551548, "sampling/sampling_logp_difference/max": 0.8395180702209473, "sampling/importance_sampling_ratio/min": 0.4319186508655548, "sampling/importance_sampling_ratio/mean": 0.9952727556228638, "sampling/importance_sampling_ratio/max": 1.063112735748291, "entropy": 0.060785280307754874, "clip_ratio/low_mean": 0.0033783784601837397, "clip_ratio/low_min": 0.0033783784601837397, "clip_ratio/high_mean": 0.0035714285913854837, "clip_ratio/high_max": 0.0035714285913854837, "clip_ratio/region_mean": 0.006949807051569223, "reward_total_mean": 0.9971714019775391, "reward_meter_mean": 0.9971714019775391, "reward_meter_std": 0.0016039953334257007, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9971714019775391, "reward_total_composite_std": 0.0016039953334257007} {"timestamp_utc": "2026-04-12T00:33:03Z", "mode": "train", "global_step": 1629, "epoch": 0.06542956982769009, "loss": -0.0053, "grad_norm": 3.441396474838257, "learning_rate": 5.0666666666666676e-06, "num_tokens": 3663869.0, "completions/mean_length": 339.125, "completions/min_length": 321.0, "completions/max_length": 360.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 339.125, "completions/min_terminated_length": 321.0, "completions/max_terminated_length": 360.0, "rewards/meter/mean": 0.9954209923744202, "rewards/meter/std": 0.0042012035846710205, "rewards/count_adherence/mean": 0.8068181276321411, "rewards/count_adherence/std": 0.03214120864868164, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9562908411026001, "rewards/repeat_penalty/std": 0.060786258429288864, "rewards/total_composite/mean": 0.7674725651741028, "rewards/total_composite/std": 0.04974528029561043, "reward": 0.7674725651741028, "reward_std": 0.04974529147148132, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06467485427856445, "sampling/sampling_logp_difference/max": 1.5753440856933594, "sampling/importance_sampling_ratio/min": 0.2069363296031952, "sampling/importance_sampling_ratio/mean": 1.0179475545883179, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.622920136898756, "clip_ratio/low_mean": 0.004980936297215521, "clip_ratio/low_min": 0.004980936297215521, "clip_ratio/high_mean": 0.03499942785128951, "clip_ratio/high_max": 0.03499942785128951, "clip_ratio/region_mean": 0.03998036414850503, "reward_total_mean": 0.7674725651741028, "reward_meter_mean": 0.9954209923744202, "reward_meter_std": 0.0042012035846710205, "reward_count_adherence_mean": 0.8068181276321411, "reward_count_adherence_std": 0.03214120864868164, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9562908411026001, "reward_repeat_penalty_std": 0.060786258429288864, "reward_total_composite_mean": 0.7674725651741028, "reward_total_composite_std": 0.04974528029561043} {"timestamp_utc": "2026-04-12T00:33:12Z", "mode": "train", "global_step": 1630, "epoch": 0.06546973530947504, "loss": 0.01, "grad_norm": 2.8746509552001953, "learning_rate": 5.063636363636364e-06, "num_tokens": 3668262.0, "completions/mean_length": 343.125, "completions/min_length": 326.0, "completions/max_length": 355.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 343.125, "completions/min_terminated_length": 326.0, "completions/max_terminated_length": 355.0, "rewards/meter/mean": 0.9573882222175598, "rewards/meter/std": 0.10380185395479202, "rewards/count_adherence/mean": 0.862500011920929, "rewards/count_adherence/std": 0.0517548993229866, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9774816036224365, "rewards/repeat_penalty/std": 0.04422420263290405, "rewards/total_composite/mean": 0.8051279783248901, "rewards/total_composite/std": 0.09284678101539612, "reward": 0.8051279783248901, "reward_std": 0.09284677356481552, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07694988697767258, "sampling/sampling_logp_difference/max": 1.7180042266845703, "sampling/importance_sampling_ratio/min": 0.17942388355731964, "sampling/importance_sampling_ratio/mean": 1.016446590423584, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7744914256036282, "clip_ratio/low_mean": 0.031975225545465946, "clip_ratio/low_min": 0.031975225545465946, "clip_ratio/high_mean": 0.02247373666614294, "clip_ratio/high_max": 0.02247373666614294, "clip_ratio/region_mean": 0.05444896221160889, "reward_total_mean": 0.8051279783248901, "reward_meter_mean": 0.9573882222175598, "reward_meter_std": 0.10380185395479202, "reward_count_adherence_mean": 0.862500011920929, "reward_count_adherence_std": 0.0517548993229866, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9774816036224365, "reward_repeat_penalty_std": 0.04422420263290405, "reward_total_composite_mean": 0.8051279783248901, "reward_total_composite_std": 0.09284678101539612} {"timestamp_utc": "2026-04-12T00:33:17Z", "mode": "train", "global_step": 1631, "epoch": 0.06550990079126, "loss": 0.0001, "grad_norm": 0.9008557200431824, "learning_rate": 5.060606060606061e-06, "num_tokens": 3670677.0, "completions/mean_length": 127.875, "completions/min_length": 127.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.875, "completions/min_terminated_length": 127.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.9939456582069397, "rewards/meter/std": 0.00018129698582924902, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.75, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5963674187660217, "rewards/total_composite/std": 0.00010878406465053558, "reward": 0.5963674187660217, "reward_std": 0.0001087786877178587, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007339623291045427, "sampling/sampling_logp_difference/max": 0.8770111799240112, "sampling/importance_sampling_ratio/min": 0.41602447628974915, "sampling/importance_sampling_ratio/mean": 1.0000206232070923, "sampling/importance_sampling_ratio/max": 1.6699923276901245, "entropy": 0.02229404100216925, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/high_mean": 0.006843626964837313, "clip_ratio/high_max": 0.006843626964837313, "clip_ratio/region_mean": 0.010749876964837313, "reward_total_mean": 0.5963674187660217, "reward_meter_mean": 0.9939456582069397, "reward_meter_std": 0.00018129698582924902, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.75, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5963674187660217, "reward_total_composite_std": 0.00010878406465053558} {"timestamp_utc": "2026-04-12T00:33:22Z", "mode": "train", "global_step": 1632, "epoch": 0.06555006627304495, "loss": -0.0158, "grad_norm": 4.735329627990723, "learning_rate": 5.057575757575758e-06, "num_tokens": 3672574.0, "completions/mean_length": 73.125, "completions/min_length": 67.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.125, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9905960559844971, "rewards/meter/std": 0.008815782144665718, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9905960559844971, "rewards/total_composite/std": 0.008815782144665718, "reward": 0.9905960559844971, "reward_std": 0.008815782144665718, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07114405184984207, "sampling/sampling_logp_difference/max": 1.1808457374572754, "sampling/importance_sampling_ratio/min": 0.3070189952850342, "sampling/importance_sampling_ratio/mean": 1.0182366371154785, "sampling/importance_sampling_ratio/max": 1.9160219430923462, "entropy": 0.7365500181913376, "clip_ratio/low_mean": 0.027231683605350554, "clip_ratio/low_min": 0.027231683605350554, "clip_ratio/high_mean": 0.04253894090652466, "clip_ratio/high_max": 0.04253894090652466, "clip_ratio/region_mean": 0.06977062451187521, "reward_total_mean": 0.9905960559844971, "reward_meter_mean": 0.9905960559844971, "reward_meter_std": 0.008815782144665718, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9905960559844971, "reward_total_composite_std": 0.008815782144665718} {"timestamp_utc": "2026-04-12T00:33:26Z", "mode": "train", "global_step": 1633, "epoch": 0.0655902317548299, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.054545454545455e-06, "num_tokens": 3674102.0, "completions/mean_length": 30.0, "completions/min_length": 30.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.0, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.992271900177002, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.992271900177002, "rewards/total_composite/std": 0.0, "reward": 0.992271900177002, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0005566918407566845, "sampling/sampling_logp_difference/max": 0.005868114531040192, "sampling/importance_sampling_ratio/min": 0.998640239238739, "sampling/importance_sampling_ratio/mean": 1.0005319118499756, "sampling/importance_sampling_ratio/max": 1.0058854818344116, "entropy": 0.004795511835254729, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.992271900177002, "reward_meter_mean": 0.992271900177002, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.992271900177002, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T00:33:31Z", "mode": "train", "global_step": 1634, "epoch": 0.06563039723661486, "loss": 0.0068, "grad_norm": 5.207538604736328, "learning_rate": 5.051515151515151e-06, "num_tokens": 3676259.0, "completions/mean_length": 108.625, "completions/min_length": 103.0, "completions/max_length": 113.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.625, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 113.0, "rewards/meter/mean": 0.9962839484214783, "rewards/meter/std": 0.0024302289821207523, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9962839484214783, "rewards/total_composite/std": 0.0024302289821207523, "reward": 0.9962839484214783, "reward_std": 0.0024302229285240173, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0648764818906784, "sampling/sampling_logp_difference/max": 1.4173831939697266, "sampling/importance_sampling_ratio/min": 0.2423473745584488, "sampling/importance_sampling_ratio/mean": 1.0127630233764648, "sampling/importance_sampling_ratio/max": 1.8952765464782715, "entropy": 0.6268335022032261, "clip_ratio/low_mean": 0.013648775406181812, "clip_ratio/low_min": 0.013648775406181812, "clip_ratio/high_mean": 0.03815131215378642, "clip_ratio/high_max": 0.03815131215378642, "clip_ratio/region_mean": 0.05180008755996823, "reward_total_mean": 0.9962839484214783, "reward_meter_mean": 0.9962839484214783, "reward_meter_std": 0.0024302289821207523, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9962839484214783, "reward_total_composite_std": 0.0024302289821207523} {"timestamp_utc": "2026-04-12T00:33:37Z", "mode": "train", "global_step": 1635, "epoch": 0.06567056271839981, "loss": -0.0025, "grad_norm": 4.150503158569336, "learning_rate": 5.048484848484849e-06, "num_tokens": 3679020.0, "completions/mean_length": 160.125, "completions/min_length": 153.0, "completions/max_length": 166.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 160.125, "completions/min_terminated_length": 153.0, "completions/max_terminated_length": 166.0, "rewards/meter/mean": 0.9920204877853394, "rewards/meter/std": 0.01226998120546341, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.974323570728302, "rewards/total_composite/std": 0.05192466080188751, "reward": 0.974323570728302, "reward_std": 0.05192466452717781, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08837065100669861, "sampling/sampling_logp_difference/max": 1.2528629302978516, "sampling/importance_sampling_ratio/min": 0.2856857180595398, "sampling/importance_sampling_ratio/mean": 1.0256816148757935, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9452630281448364, "clip_ratio/low_mean": 0.01331305643543601, "clip_ratio/low_min": 0.01331305643543601, "clip_ratio/high_mean": 0.042927493108436465, "clip_ratio/high_max": 0.042927493108436465, "clip_ratio/region_mean": 0.056240549543872476, "reward_total_mean": 0.974323570728302, "reward_meter_mean": 0.9920204877853394, "reward_meter_std": 0.01226998120546341, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.974323570728302, "reward_total_composite_std": 0.05192466080188751} {"timestamp_utc": "2026-04-12T00:33:42Z", "mode": "train", "global_step": 1636, "epoch": 0.06571072820018477, "loss": 0.0004, "grad_norm": 2.887805223464966, "learning_rate": 5.045454545454546e-06, "num_tokens": 3680811.0, "completions/mean_length": 63.875, "completions/min_length": 63.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.875, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9980711936950684, "rewards/meter/std": 0.0005036595975980163, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980711936950684, "rewards/total_composite/std": 0.0005036595975980163, "reward": 0.9980711936950684, "reward_std": 0.0005036554648540914, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019482869654893875, "sampling/sampling_logp_difference/max": 0.7397842407226562, "sampling/importance_sampling_ratio/min": 0.477216899394989, "sampling/importance_sampling_ratio/mean": 0.9980214238166809, "sampling/importance_sampling_ratio/max": 1.4901469945907593, "entropy": 0.09926960244774818, "clip_ratio/low_mean": 0.007905506063252687, "clip_ratio/low_min": 0.007905506063252687, "clip_ratio/high_mean": 0.00390625, "clip_ratio/high_max": 0.00390625, "clip_ratio/region_mean": 0.011811756063252687, "reward_total_mean": 0.9980711936950684, "reward_meter_mean": 0.9980711936950684, "reward_meter_std": 0.0005036595975980163, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9980711936950684, "reward_total_composite_std": 0.0005036595975980163} {"timestamp_utc": "2026-04-12T00:33:48Z", "mode": "train", "global_step": 1637, "epoch": 0.06575089368196972, "loss": 0.0281, "grad_norm": 7.507004737854004, "learning_rate": 5.042424242424243e-06, "num_tokens": 3682330.0, "completions/mean_length": 69.875, "completions/min_length": 63.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.875, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.6862971782684326, "rewards/meter/std": 0.2807905673980713, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6862971782684326, "rewards/total_composite/std": 0.2807905673980713, "reward": 0.6862971782684326, "reward_std": 0.2807905673980713, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10850861668586731, "sampling/sampling_logp_difference/max": 1.5064477920532227, "sampling/importance_sampling_ratio/min": 0.22169609367847443, "sampling/importance_sampling_ratio/mean": 1.0039104223251343, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0178619101643562, "clip_ratio/low_mean": 0.026575686410069466, "clip_ratio/low_min": 0.026575686410069466, "clip_ratio/high_mean": 0.05837191268801689, "clip_ratio/high_max": 0.05837191268801689, "clip_ratio/region_mean": 0.08494759909808636, "reward_total_mean": 0.6862971782684326, "reward_meter_mean": 0.6862971782684326, "reward_meter_std": 0.2807905673980713, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6862971782684326, "reward_total_composite_std": 0.2807905673980713} {"timestamp_utc": "2026-04-12T00:33:53Z", "mode": "train", "global_step": 1638, "epoch": 0.06579105916375468, "loss": 0.0097, "grad_norm": 3.420053720474243, "learning_rate": 5.0393939393939395e-06, "num_tokens": 3684458.0, "completions/mean_length": 95.0, "completions/min_length": 94.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.0, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9959671497344971, "rewards/meter/std": 0.006246180739253759, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.8965927362442017, "rewards/total_composite/std": 0.1087099239230156, "reward": 0.8965927362442017, "reward_std": 0.10870993137359619, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01706601306796074, "sampling/sampling_logp_difference/max": 0.9803454875946045, "sampling/importance_sampling_ratio/min": 0.6205868124961853, "sampling/importance_sampling_ratio/mean": 1.0042842626571655, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10523929260671139, "clip_ratio/low_mean": 0.006431827903725207, "clip_ratio/low_min": 0.006431827903725207, "clip_ratio/high_mean": 0.010610593715682626, "clip_ratio/high_max": 0.010610593715682626, "clip_ratio/region_mean": 0.017042421619407833, "reward_total_mean": 0.8965927362442017, "reward_meter_mean": 0.9959671497344971, "reward_meter_std": 0.006246180739253759, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_total_composite_mean": 0.8965927362442017, "reward_total_composite_std": 0.1087099239230156} {"timestamp_utc": "2026-04-12T00:33:58Z", "mode": "train", "global_step": 1639, "epoch": 0.06583122464553963, "loss": 0.0003, "grad_norm": 1.6942023038864136, "learning_rate": 5.036363636363637e-06, "num_tokens": 3686354.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9983768463134766, "rewards/meter/std": 0.00010835672583198175, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9983768463134766, "rewards/total_composite/std": 0.00010835672583198175, "reward": 0.9983768463134766, "reward_std": 0.00010834616841748357, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01173328422009945, "sampling/sampling_logp_difference/max": 0.8327422142028809, "sampling/importance_sampling_ratio/min": 0.434855192899704, "sampling/importance_sampling_ratio/mean": 1.0026438236236572, "sampling/importance_sampling_ratio/max": 1.363644003868103, "entropy": 0.06925144325941801, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/high_mean": 0.005859375, "clip_ratio/high_max": 0.005859375, "clip_ratio/region_mean": 0.009765625, "reward_total_mean": 0.9983768463134766, "reward_meter_mean": 0.9983768463134766, "reward_meter_std": 0.00010835672583198175, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9983768463134766, "reward_total_composite_std": 0.00010835672583198175} {"timestamp_utc": "2026-04-12T00:34:03Z", "mode": "train", "global_step": 1640, "epoch": 0.06587139012732458, "loss": 0.0239, "grad_norm": 5.066521167755127, "learning_rate": 5.033333333333333e-06, "num_tokens": 3688136.0, "completions/mean_length": 69.75, "completions/min_length": 67.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.75, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9961869716644287, "rewards/meter/std": 0.0016419171588495374, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9961869716644287, "rewards/total_composite/std": 0.0016419171588495374, "reward": 0.9961869716644287, "reward_std": 0.0016419151797890663, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.036743052303791046, "sampling/sampling_logp_difference/max": 0.8399066925048828, "sampling/importance_sampling_ratio/min": 0.4317508041858673, "sampling/importance_sampling_ratio/mean": 1.0081746578216553, "sampling/importance_sampling_ratio/max": 1.939770221710205, "entropy": 0.2406612578779459, "clip_ratio/low_mean": 0.010499144555069506, "clip_ratio/low_min": 0.010499144555069506, "clip_ratio/high_mean": 0.012254902278073132, "clip_ratio/high_max": 0.012254902278073132, "clip_ratio/region_mean": 0.022754046833142638, "reward_total_mean": 0.9961869716644287, "reward_meter_mean": 0.9961869716644287, "reward_meter_std": 0.0016419171588495374, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9961869716644287, "reward_total_composite_std": 0.0016419171588495374} {"timestamp_utc": "2026-04-12T00:34:07Z", "mode": "train", "global_step": 1641, "epoch": 0.06591155560910954, "loss": -0.0356, "grad_norm": 6.707535266876221, "learning_rate": 5.030303030303031e-06, "num_tokens": 3690011.0, "completions/mean_length": 69.375, "completions/min_length": 61.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.375, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.7832512855529785, "rewards/meter/std": 0.33663251996040344, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7832512855529785, "rewards/total_composite/std": 0.33663251996040344, "reward": 0.7832512855529785, "reward_std": 0.33663249015808105, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09360064566135406, "sampling/sampling_logp_difference/max": 1.031529426574707, "sampling/importance_sampling_ratio/min": 0.35646137595176697, "sampling/importance_sampling_ratio/mean": 1.0248838663101196, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9592136070132256, "clip_ratio/low_mean": 0.01592771988362074, "clip_ratio/low_min": 0.01592771988362074, "clip_ratio/high_mean": 0.047490152064710855, "clip_ratio/high_max": 0.047490152064710855, "clip_ratio/region_mean": 0.0634178719483316, "reward_total_mean": 0.7832512855529785, "reward_meter_mean": 0.7832512855529785, "reward_meter_std": 0.33663251996040344, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7832512855529785, "reward_total_composite_std": 0.33663251996040344} {"timestamp_utc": "2026-04-12T00:34:12Z", "mode": "train", "global_step": 1642, "epoch": 0.06595172109089449, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.027272727272728e-06, "num_tokens": 3691659.0, "completions/mean_length": 58.0, "completions/min_length": 58.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.993520200252533, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.993520200252533, "rewards/total_composite/std": 0.0, "reward": 0.993520200252533, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0005443372065201402, "sampling/sampling_logp_difference/max": 0.014515344053506851, "sampling/importance_sampling_ratio/min": 0.9973773956298828, "sampling/importance_sampling_ratio/mean": 1.0005112886428833, "sampling/importance_sampling_ratio/max": 1.0146211385726929, "entropy": 0.0062332607340067625, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.993520200252533, "reward_meter_mean": 0.993520200252533, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.993520200252533, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T00:34:18Z", "mode": "train", "global_step": 1643, "epoch": 0.06599188657267945, "loss": 0.0248, "grad_norm": 3.824038028717041, "learning_rate": 5.024242424242425e-06, "num_tokens": 3695039.0, "completions/mean_length": 203.5, "completions/min_length": 193.0, "completions/max_length": 227.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 203.5, "completions/min_terminated_length": 193.0, "completions/max_terminated_length": 227.0, "rewards/meter/mean": 0.9925307035446167, "rewards/meter/std": 0.010234549641609192, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9925307035446167, "rewards/total_composite/std": 0.010234549641609192, "reward": 0.9925307035446167, "reward_std": 0.010234540328383446, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07599979639053345, "sampling/sampling_logp_difference/max": 1.208749771118164, "sampling/importance_sampling_ratio/min": 0.29857030510902405, "sampling/importance_sampling_ratio/mean": 1.0239778757095337, "sampling/importance_sampling_ratio/max": 1.9295165538787842, "entropy": 0.8424399569630623, "clip_ratio/low_mean": 0.009563126135617495, "clip_ratio/low_min": 0.009563126135617495, "clip_ratio/high_mean": 0.042821857146918774, "clip_ratio/high_max": 0.042821857146918774, "clip_ratio/region_mean": 0.05238498328253627, "reward_total_mean": 0.9925307035446167, "reward_meter_mean": 0.9925307035446167, "reward_meter_std": 0.010234549641609192, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9925307035446167, "reward_total_composite_std": 0.010234549641609192} {"timestamp_utc": "2026-04-12T00:34:24Z", "mode": "train", "global_step": 1644, "epoch": 0.0660320520544644, "loss": -0.0099, "grad_norm": 3.962242603302002, "learning_rate": 5.021212121212121e-06, "num_tokens": 3697742.0, "completions/mean_length": 146.875, "completions/min_length": 142.0, "completions/max_length": 152.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 146.875, "completions/min_terminated_length": 142.0, "completions/max_terminated_length": 152.0, "rewards/meter/mean": 0.9914488792419434, "rewards/meter/std": 0.019252995029091835, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9379702806472778, "rewards/total_composite/std": 0.07060353457927704, "reward": 0.9379702806472778, "reward_std": 0.07060354202985764, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0685059130191803, "sampling/sampling_logp_difference/max": 1.3913602828979492, "sampling/importance_sampling_ratio/min": 0.24873672425746918, "sampling/importance_sampling_ratio/mean": 1.0191352367401123, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6761485747992992, "clip_ratio/low_mean": 0.016500307247042656, "clip_ratio/low_min": 0.016500307247042656, "clip_ratio/high_mean": 0.037598957773298025, "clip_ratio/high_max": 0.037598957773298025, "clip_ratio/region_mean": 0.05409926502034068, "reward_total_mean": 0.9379702806472778, "reward_meter_mean": 0.9914488792419434, "reward_meter_std": 0.019252995029091835, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.9379702806472778, "reward_total_composite_std": 0.07060353457927704} {"timestamp_utc": "2026-04-12T00:34:29Z", "mode": "train", "global_step": 1645, "epoch": 0.06607221753624935, "loss": -0.0102, "grad_norm": 7.402265548706055, "learning_rate": 5.0181818181818186e-06, "num_tokens": 3699539.0, "completions/mean_length": 69.625, "completions/min_length": 62.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.625, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.6198569536209106, "rewards/meter/std": 0.3495922088623047, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6198569536209106, "rewards/total_composite/std": 0.3495922088623047, "reward": 0.6198569536209106, "reward_std": 0.3495922088623047, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10739893466234207, "sampling/sampling_logp_difference/max": 1.626225471496582, "sampling/importance_sampling_ratio/min": 0.1966705024242401, "sampling/importance_sampling_ratio/mean": 1.0240029096603394, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.080611765384674, "clip_ratio/low_mean": 0.022980431327596307, "clip_ratio/low_min": 0.022980431327596307, "clip_ratio/high_mean": 0.053518022410571575, "clip_ratio/high_max": 0.053518022410571575, "clip_ratio/region_mean": 0.07649845373816788, "reward_total_mean": 0.6198569536209106, "reward_meter_mean": 0.6198569536209106, "reward_meter_std": 0.3495922088623047, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6198569536209106, "reward_total_composite_std": 0.3495922088623047} {"timestamp_utc": "2026-04-12T00:34:33Z", "mode": "train", "global_step": 1646, "epoch": 0.06611238301803431, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.015151515151515e-06, "num_tokens": 3701219.0, "completions/mean_length": 58.0, "completions/min_length": 58.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.993520200252533, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.993520200252533, "rewards/total_composite/std": 0.0, "reward": 0.993520200252533, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0016902285860851407, "sampling/sampling_logp_difference/max": 0.07047711312770844, "sampling/importance_sampling_ratio/min": 0.9319490194320679, "sampling/importance_sampling_ratio/mean": 1.0011212825775146, "sampling/importance_sampling_ratio/max": 1.0615943670272827, "entropy": 0.010349510470405221, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.993520200252533, "reward_meter_mean": 0.993520200252533, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.993520200252533, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T00:34:38Z", "mode": "train", "global_step": 1647, "epoch": 0.06615254849981926, "loss": 0.0056, "grad_norm": 4.599454879760742, "learning_rate": 5.012121212121212e-06, "num_tokens": 3703551.0, "completions/mean_length": 115.5, "completions/min_length": 110.0, "completions/max_length": 120.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.5, "completions/min_terminated_length": 110.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.989443302154541, "rewards/meter/std": 0.009337217546999454, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.989443302154541, "rewards/total_composite/std": 0.009337217546999454, "reward": 0.989443302154541, "reward_std": 0.009337219409644604, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07534082978963852, "sampling/sampling_logp_difference/max": 1.2979240417480469, "sampling/importance_sampling_ratio/min": 0.2730981409549713, "sampling/importance_sampling_ratio/mean": 1.0189144611358643, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7444454878568649, "clip_ratio/low_mean": 0.02626075316220522, "clip_ratio/low_min": 0.02626075316220522, "clip_ratio/high_mean": 0.025925687979906797, "clip_ratio/high_max": 0.025925687979906797, "clip_ratio/region_mean": 0.05218644114211202, "reward_total_mean": 0.989443302154541, "reward_meter_mean": 0.989443302154541, "reward_meter_std": 0.009337217546999454, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.989443302154541, "reward_total_composite_std": 0.009337217546999454} {"timestamp_utc": "2026-04-12T00:34:43Z", "mode": "train", "global_step": 1648, "epoch": 0.06619271398160421, "loss": 0.0453, "grad_norm": 5.8860039710998535, "learning_rate": 5.009090909090909e-06, "num_tokens": 3705642.0, "completions/mean_length": 103.375, "completions/min_length": 97.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.375, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.877294659614563, "rewards/meter/std": 0.19824109971523285, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.877294659614563, "rewards/total_composite/std": 0.19824109971523285, "reward": 0.877294659614563, "reward_std": 0.19824109971523285, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10711432248353958, "sampling/sampling_logp_difference/max": 1.7170472145080566, "sampling/importance_sampling_ratio/min": 0.1795956790447235, "sampling/importance_sampling_ratio/mean": 1.031799077987671, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.183668315410614, "clip_ratio/low_mean": 0.01907266629859805, "clip_ratio/low_min": 0.01907266629859805, "clip_ratio/high_mean": 0.06016605906188488, "clip_ratio/high_max": 0.06016605906188488, "clip_ratio/region_mean": 0.07923872536048293, "reward_total_mean": 0.877294659614563, "reward_meter_mean": 0.877294659614563, "reward_meter_std": 0.19824109971523285, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.877294659614563, "reward_total_composite_std": 0.19824109971523285} {"timestamp_utc": "2026-04-12T00:34:48Z", "mode": "train", "global_step": 1649, "epoch": 0.06623287946338917, "loss": -0.0025, "grad_norm": 5.83959436416626, "learning_rate": 5.006060606060607e-06, "num_tokens": 3707194.0, "completions/mean_length": 32.0, "completions/min_length": 31.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9442778825759888, "rewards/meter/std": 0.059931062161922455, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9442778825759888, "rewards/total_composite/std": 0.059931062161922455, "reward": 0.9442778825759888, "reward_std": 0.059931062161922455, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05596788227558136, "sampling/sampling_logp_difference/max": 1.9147157669067383, "sampling/importance_sampling_ratio/min": 0.14738371968269348, "sampling/importance_sampling_ratio/mean": 1.007331371307373, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4391027558594942, "clip_ratio/low_mean": 0.008064515888690948, "clip_ratio/low_min": 0.008064515888690948, "clip_ratio/high_mean": 0.03531280532479286, "clip_ratio/high_max": 0.03531280532479286, "clip_ratio/region_mean": 0.04337732121348381, "reward_total_mean": 0.9442778825759888, "reward_meter_mean": 0.9442778825759888, "reward_meter_std": 0.059931062161922455, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9442778825759888, "reward_total_composite_std": 0.059931062161922455} {"timestamp_utc": "2026-04-12T00:34:52Z", "mode": "train", "global_step": 1650, "epoch": 0.06627304494517412, "loss": -0.0102, "grad_norm": 4.691352367401123, "learning_rate": 5.003030303030303e-06, "num_tokens": 3708819.0, "completions/mean_length": 32.125, "completions/min_length": 31.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.125, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9763838648796082, "rewards/meter/std": 0.04593488201498985, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9763838648796082, "rewards/total_composite/std": 0.04593488201498985, "reward": 0.9763838648796082, "reward_std": 0.04593489319086075, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05742637813091278, "sampling/sampling_logp_difference/max": 1.3233098983764648, "sampling/importance_sampling_ratio/min": 0.2662525773048401, "sampling/importance_sampling_ratio/mean": 1.0103769302368164, "sampling/importance_sampling_ratio/max": 1.4046815633773804, "entropy": 0.44733541272580624, "clip_ratio/low_mean": 0.004032257944345474, "clip_ratio/low_min": 0.004032257944345474, "clip_ratio/high_mean": 0.0337981628254056, "clip_ratio/high_max": 0.0337981628254056, "clip_ratio/region_mean": 0.03783042076975107, "reward_total_mean": 0.9763838648796082, "reward_meter_mean": 0.9763838648796082, "reward_meter_std": 0.04593488201498985, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9763838648796082, "reward_total_composite_std": 0.04593488201498985} {"timestamp_utc": "2026-04-12T00:36:11Z", "mode": "eval", "global_step": 1650, "epoch": 0.06627304494517412, "eval_loss": NaN, "eval_runtime": 79.2708, "eval_samples_per_second": 1.312, "eval_steps_per_second": 0.164, "eval_num_tokens": 3708819.0, "eval_completions/mean_length": 212.91346153846155, "eval_completions/min_length": 62.0, "eval_completions/max_length": 409.3076923076923, "eval_completions/clipped_ratio": 0.028846153846153848, "eval_completions/mean_terminated_length": 204.67307927058295, "eval_completions/min_terminated_length": 62.0, "eval_completions/max_terminated_length": 372.9230769230769, "eval_rewards/meter/mean": 0.7244124962733343, "eval_rewards/meter/std": 0.36815990120745623, "eval_rewards/count_adherence/mean": 0.9108829039793748, "eval_rewards/count_adherence/std": 0.11557848694232795, "eval_rewards/arabic_clean/mean": 0.9807692307692307, "eval_rewards/arabic_clean/std": 0.05439282839114849, "eval_rewards/repeat_penalty/mean": 0.8433380264502305, "eval_rewards/repeat_penalty/std": 0.15947996452450752, "eval_rewards/total_composite/mean": 0.5518424786054171, "eval_rewards/total_composite/std": 0.34955332485529095, "eval_reward": 0.5518424786054171, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.035442069316139586, "eval_sampling/sampling_logp_difference/max": 1.0809908371705275, "eval_sampling/importance_sampling_ratio/min": 0.3488806807077848, "eval_sampling/importance_sampling_ratio/mean": 1.009941733800448, "eval_sampling/importance_sampling_ratio/max": 1.5815464166494517, "eval_entropy": 0.4149218121400246, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5518424786054171, "eval_reward_meter_mean": 0.7244124962733343, "eval_reward_meter_std": 0.36815990120745623, "eval_reward_count_adherence_mean": 0.9108829039793748, "eval_reward_count_adherence_std": 0.11557848694232795, "eval_reward_arabic_clean_mean": 0.9807692307692307, "eval_reward_arabic_clean_std": 0.05439282839114849, "eval_reward_repeat_penalty_mean": 0.8433380264502305, "eval_reward_repeat_penalty_std": 0.15947996452450752, "eval_reward_total_composite_mean": 0.5518424786054171, "eval_reward_total_composite_std": 0.34955332485529095} {"timestamp_utc": "2026-04-12T00:36:18Z", "mode": "train", "global_step": 1651, "epoch": 0.06631321042695908, "loss": 0.0001, "grad_norm": 0.5019415616989136, "learning_rate": 5e-06, "num_tokens": 3710592.0, "completions/mean_length": 67.625, "completions/min_length": 67.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.625, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9974769949913025, "rewards/meter/std": 0.001003190758638084, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9974769949913025, "rewards/total_composite/std": 0.001003190758638084, "reward": 0.9974769949913025, "reward_std": 0.001003190758638084, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009622362442314625, "sampling/sampling_logp_difference/max": 0.6310257911682129, "sampling/importance_sampling_ratio/min": 0.5320457816123962, "sampling/importance_sampling_ratio/mean": 1.0023994445800781, "sampling/importance_sampling_ratio/max": 1.4636121988296509, "entropy": 0.06585087720304728, "clip_ratio/low_mean": 0.0018382353009656072, "clip_ratio/low_min": 0.0018382353009656072, "clip_ratio/high_mean": 0.005569578497670591, "clip_ratio/high_max": 0.005569578497670591, "clip_ratio/region_mean": 0.007407813798636198, "reward_total_mean": 0.9974769949913025, "reward_meter_mean": 0.9974769949913025, "reward_meter_std": 0.001003190758638084, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9974769949913025, "reward_total_composite_std": 0.001003190758638084} {"timestamp_utc": "2026-04-12T00:36:23Z", "mode": "train", "global_step": 1652, "epoch": 0.06635337590874403, "loss": -0.0038, "grad_norm": 1.1626874208450317, "learning_rate": 4.996969696969698e-06, "num_tokens": 3712383.0, "completions/mean_length": 67.875, "completions/min_length": 67.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.875, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9978716373443604, "rewards/meter/std": 7.145507697714493e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978716373443604, "rewards/total_composite/std": 7.145507697714493e-05, "reward": 0.9978716373443604, "reward_std": 7.145912968553603e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011869369074702263, "sampling/sampling_logp_difference/max": 0.9046740531921387, "sampling/importance_sampling_ratio/min": 0.4046737849712372, "sampling/importance_sampling_ratio/mean": 1.0000650882720947, "sampling/importance_sampling_ratio/max": 1.2595540285110474, "entropy": 0.07074726792052388, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.007299659075215459, "clip_ratio/high_max": 0.007299659075215459, "clip_ratio/region_mean": 0.007299659075215459, "reward_total_mean": 0.9978716373443604, "reward_meter_mean": 0.9978716373443604, "reward_meter_std": 7.145507697714493e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9978716373443604, "reward_total_composite_std": 7.145507697714493e-05} {"timestamp_utc": "2026-04-12T00:36:34Z", "mode": "train", "global_step": 1653, "epoch": 0.06639354139052898, "loss": -0.3504, "grad_norm": 1.3737280368804932, "learning_rate": 4.993939393939394e-06, "num_tokens": 3716159.0, "completions/mean_length": 382.0, "completions/min_length": 321.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 338.66668701171875, "completions/min_terminated_length": 321.0, "completions/max_terminated_length": 361.0, "rewards/meter/mean": 0.8152369260787964, "rewards/meter/std": 0.1942194551229477, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 0.9428104758262634, "rewards/repeat_penalty/std": 0.05349903926253319, "rewards/total_composite/mean": 0.5758238434791565, "rewards/total_composite/std": 0.36915671825408936, "reward": 0.5758238434791565, "reward_std": 0.36915671825408936, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08613347262144089, "sampling/sampling_logp_difference/max": 2.176515579223633, "sampling/importance_sampling_ratio/min": 0.11343610286712646, "sampling/importance_sampling_ratio/mean": 1.0216100215911865, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7133785486221313, "clip_ratio/low_mean": 0.007788162212818861, "clip_ratio/low_min": 0.007788162212818861, "clip_ratio/high_mean": 0.03797963773831725, "clip_ratio/high_max": 0.03797963773831725, "clip_ratio/region_mean": 0.04576779995113611, "reward_total_mean": 0.5758238434791565, "reward_meter_mean": 0.8152369260787964, "reward_meter_std": 0.1942194551229477, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 0.9428104758262634, "reward_repeat_penalty_std": 0.05349903926253319, "reward_total_composite_mean": 0.5758238434791565, "reward_total_composite_std": 0.36915671825408936} {"timestamp_utc": "2026-04-12T00:36:39Z", "mode": "train", "global_step": 1654, "epoch": 0.06643370687231394, "loss": -0.0232, "grad_norm": 3.5595715045928955, "learning_rate": 4.990909090909091e-06, "num_tokens": 3718004.0, "completions/mean_length": 63.625, "completions/min_length": 62.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.625, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9683968424797058, "rewards/meter/std": 0.024647340178489685, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9683968424797058, "rewards/total_composite/std": 0.024647340178489685, "reward": 0.9683968424797058, "reward_std": 0.024647342041134834, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03070976585149765, "sampling/sampling_logp_difference/max": 5.06722354888916, "sampling/importance_sampling_ratio/min": 0.00629988731816411, "sampling/importance_sampling_ratio/mean": 0.9969826936721802, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0451802066527307, "clip_ratio/low_mean": 0.010080644860863686, "clip_ratio/low_min": 0.010080644860863686, "clip_ratio/high_mean": 0.00562528264708817, "clip_ratio/high_max": 0.00562528264708817, "clip_ratio/region_mean": 0.015705927507951856, "reward_total_mean": 0.9683968424797058, "reward_meter_mean": 0.9683968424797058, "reward_meter_std": 0.024647340178489685, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9683968424797058, "reward_total_composite_std": 0.024647340178489685} {"timestamp_utc": "2026-04-12T00:36:44Z", "mode": "train", "global_step": 1655, "epoch": 0.06647387235409889, "loss": -0.0142, "grad_norm": 5.8942670822143555, "learning_rate": 4.987878787878789e-06, "num_tokens": 3719438.0, "completions/mean_length": 33.25, "completions/min_length": 32.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.25, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9977890849113464, "rewards/meter/std": 0.002695126226171851, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977890849113464, "rewards/total_composite/std": 0.002695126226171851, "reward": 0.9977890849113464, "reward_std": 0.002695116912946105, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0360952690243721, "sampling/sampling_logp_difference/max": 4.130129814147949, "sampling/importance_sampling_ratio/min": 0.016080791130661964, "sampling/importance_sampling_ratio/mean": 0.9885497093200684, "sampling/importance_sampling_ratio/max": 1.4004013538360596, "entropy": 0.03417817200534046, "clip_ratio/low_mean": 0.0078125, "clip_ratio/low_min": 0.0078125, "clip_ratio/high_mean": 0.014928699005395174, "clip_ratio/high_max": 0.014928699005395174, "clip_ratio/region_mean": 0.022741199005395174, "reward_total_mean": 0.9977890849113464, "reward_meter_mean": 0.9977890849113464, "reward_meter_std": 0.002695126226171851, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9977890849113464, "reward_total_composite_std": 0.002695126226171851} {"timestamp_utc": "2026-04-12T00:36:54Z", "mode": "train", "global_step": 1656, "epoch": 0.06651403783588385, "loss": -0.2605, "grad_norm": 1.0378773212432861, "learning_rate": 4.984848484848485e-06, "num_tokens": 3722366.0, "completions/mean_length": 235.0, "completions/min_length": 182.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 195.42857360839844, "completions/min_terminated_length": 182.0, "completions/max_terminated_length": 214.0, "rewards/meter/mean": 0.8787662982940674, "rewards/meter/std": 0.3354223370552063, "rewards/count_adherence/mean": 0.925000011920929, "rewards/count_adherence/std": 0.2121320217847824, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.9861111044883728, "rewards/repeat_penalty/std": 0.03928370773792267, "rewards/total_composite/mean": 0.8588160872459412, "rewards/total_composite/std": 0.34913378953933716, "reward": 0.8588160872459412, "reward_std": 0.34913378953933716, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0746423751115799, "sampling/sampling_logp_difference/max": 1.7693357467651367, "sampling/importance_sampling_ratio/min": 0.17044617235660553, "sampling/importance_sampling_ratio/mean": 1.0213563442230225, "sampling/importance_sampling_ratio/max": 1.8245017528533936, "entropy": 0.6871326193213463, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.046408120542764664, "clip_ratio/high_max": 0.046408120542764664, "clip_ratio/region_mean": 0.046408120542764664, "reward_total_mean": 0.8588160872459412, "reward_meter_mean": 0.8787662982940674, "reward_meter_std": 0.3354223370552063, "reward_count_adherence_mean": 0.925000011920929, "reward_count_adherence_std": 0.2121320217847824, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.9861111044883728, "reward_repeat_penalty_std": 0.03928370773792267, "reward_total_composite_mean": 0.8588160872459412, "reward_total_composite_std": 0.34913378953933716} {"timestamp_utc": "2026-04-12T00:36:58Z", "mode": "train", "global_step": 1657, "epoch": 0.0665542033176688, "loss": -0.0037, "grad_norm": 6.6080169677734375, "learning_rate": 4.981818181818182e-06, "num_tokens": 3724136.0, "completions/mean_length": 63.25, "completions/min_length": 62.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.25, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9078664779663086, "rewards/meter/std": 0.15767432749271393, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9078664779663086, "rewards/total_composite/std": 0.15767432749271393, "reward": 0.9078664779663086, "reward_std": 0.15767431259155273, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.043513912707567215, "sampling/sampling_logp_difference/max": 1.9472687244415283, "sampling/importance_sampling_ratio/min": 0.14266319572925568, "sampling/importance_sampling_ratio/mean": 0.998699963092804, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3440733216702938, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.021861399058252573, "clip_ratio/high_max": 0.021861399058252573, "clip_ratio/region_mean": 0.021861399058252573, "reward_total_mean": 0.9078664779663086, "reward_meter_mean": 0.9078664779663086, "reward_meter_std": 0.15767432749271393, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9078664779663086, "reward_total_composite_std": 0.15767432749271393} {"timestamp_utc": "2026-04-12T00:37:03Z", "mode": "train", "global_step": 1658, "epoch": 0.06659436879945375, "loss": 0.0035, "grad_norm": 2.866943597793579, "learning_rate": 4.978787878787879e-06, "num_tokens": 3726020.0, "completions/mean_length": 64.5, "completions/min_length": 64.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.5, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9981748461723328, "rewards/meter/std": 7.74198560975492e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981748461723328, "rewards/total_composite/std": 7.74198560975492e-05, "reward": 0.9981748461723328, "reward_std": 7.74198560975492e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007607936859130859, "sampling/sampling_logp_difference/max": 1.5065312385559082, "sampling/importance_sampling_ratio/min": 0.22167760133743286, "sampling/importance_sampling_ratio/mean": 0.9998639822006226, "sampling/importance_sampling_ratio/max": 1.3977038860321045, "entropy": 0.02129516121931374, "clip_ratio/low_mean": 0.001923076924867928, "clip_ratio/low_min": 0.001923076924867928, "clip_ratio/high_mean": 0.001953125, "clip_ratio/high_max": 0.001953125, "clip_ratio/region_mean": 0.003876201924867928, "reward_total_mean": 0.9981748461723328, "reward_meter_mean": 0.9981748461723328, "reward_meter_std": 7.74198560975492e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981748461723328, "reward_total_composite_std": 7.74198560975492e-05} {"timestamp_utc": "2026-04-12T00:37:08Z", "mode": "train", "global_step": 1659, "epoch": 0.06663453428123871, "loss": 0.0129, "grad_norm": 5.0424885749816895, "learning_rate": 4.975757575757576e-06, "num_tokens": 3727828.0, "completions/mean_length": 68.0, "completions/min_length": 67.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9952208995819092, "rewards/meter/std": 0.004935278091579676, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9952208995819092, "rewards/total_composite/std": 0.004935278091579676, "reward": 0.9952208995819092, "reward_std": 0.004935278091579676, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009829939343035221, "sampling/sampling_logp_difference/max": 0.8581647872924805, "sampling/importance_sampling_ratio/min": 0.42393937706947327, "sampling/importance_sampling_ratio/mean": 1.0028293132781982, "sampling/importance_sampling_ratio/max": 1.3186924457550049, "entropy": 0.06244664313271642, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.005542142200283706, "clip_ratio/high_max": 0.005542142200283706, "clip_ratio/region_mean": 0.005542142200283706, "reward_total_mean": 0.9952208995819092, "reward_meter_mean": 0.9952208995819092, "reward_meter_std": 0.004935278091579676, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9952208995819092, "reward_total_composite_std": 0.004935278091579676} {"timestamp_utc": "2026-04-12T00:37:14Z", "mode": "train", "global_step": 1660, "epoch": 0.06667469976302366, "loss": -0.0015, "grad_norm": 0.4592190384864807, "learning_rate": 4.972727272727273e-06, "num_tokens": 3730317.0, "completions/mean_length": 139.125, "completions/min_length": 139.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 139.125, "completions/min_terminated_length": 139.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.9948010444641113, "rewards/meter/std": 0.004489070735871792, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.75, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7461007833480835, "rewards/total_composite/std": 0.003366807708516717, "reward": 0.7461007833480835, "reward_std": 0.0033668053802102804, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004387721884995699, "sampling/sampling_logp_difference/max": 1.0749011039733887, "sampling/importance_sampling_ratio/min": 0.34133151173591614, "sampling/importance_sampling_ratio/mean": 0.9996078014373779, "sampling/importance_sampling_ratio/max": 1.1272794008255005, "entropy": 0.016455981065519154, "clip_ratio/low_mean": 0.0008992805960588157, "clip_ratio/low_min": 0.0008992805960588157, "clip_ratio/high_mean": 0.0017985611921176314, "clip_ratio/high_max": 0.0017985611921176314, "clip_ratio/region_mean": 0.002697841788176447, "reward_total_mean": 0.7461007833480835, "reward_meter_mean": 0.9948010444641113, "reward_meter_std": 0.004489070735871792, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.75, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7461007833480835, "reward_total_composite_std": 0.003366807708516717} {"timestamp_utc": "2026-04-12T00:37:19Z", "mode": "train", "global_step": 1661, "epoch": 0.06671486524480862, "loss": 0.0154, "grad_norm": 5.378631114959717, "learning_rate": 4.9696969696969696e-06, "num_tokens": 3732140.0, "completions/mean_length": 58.875, "completions/min_length": 58.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.875, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9912177324295044, "rewards/meter/std": 0.010196500457823277, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9912177324295044, "rewards/total_composite/std": 0.010196500457823277, "reward": 0.9912177324295044, "reward_std": 0.010196508839726448, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015276388265192509, "sampling/sampling_logp_difference/max": 0.687659740447998, "sampling/importance_sampling_ratio/min": 0.5027512907981873, "sampling/importance_sampling_ratio/mean": 1.0054762363433838, "sampling/importance_sampling_ratio/max": 1.3040835857391357, "entropy": 0.14373003505170345, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.008511104620993137, "clip_ratio/high_max": 0.008511104620993137, "clip_ratio/region_mean": 0.008511104620993137, "reward_total_mean": 0.9912177324295044, "reward_meter_mean": 0.9912177324295044, "reward_meter_std": 0.010196500457823277, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9912177324295044, "reward_total_composite_std": 0.010196500457823277} {"timestamp_utc": "2026-04-12T00:37:23Z", "mode": "train", "global_step": 1662, "epoch": 0.06675503072659357, "loss": 0.0131, "grad_norm": 3.579711675643921, "learning_rate": 4.966666666666667e-06, "num_tokens": 3733822.0, "completions/mean_length": 59.25, "completions/min_length": 58.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.25, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9940330982208252, "rewards/meter/std": 0.002611657604575157, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9940330982208252, "rewards/total_composite/std": 0.002611657604575157, "reward": 0.9940330982208252, "reward_std": 0.002611662959679961, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015829751268029213, "sampling/sampling_logp_difference/max": 0.7537941932678223, "sampling/importance_sampling_ratio/min": 0.4705776870250702, "sampling/importance_sampling_ratio/mean": 1.0035570859909058, "sampling/importance_sampling_ratio/max": 1.5864135026931763, "entropy": 0.1442255014553666, "clip_ratio/low_mean": 0.008064515888690948, "clip_ratio/low_min": 0.008064515888690948, "clip_ratio/high_mean": 0.008474576286971569, "clip_ratio/high_max": 0.008474576286971569, "clip_ratio/region_mean": 0.016539092175662518, "reward_total_mean": 0.9940330982208252, "reward_meter_mean": 0.9940330982208252, "reward_meter_std": 0.002611657604575157, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9940330982208252, "reward_total_composite_std": 0.002611657604575157} {"timestamp_utc": "2026-04-12T00:37:28Z", "mode": "train", "global_step": 1663, "epoch": 0.06679519620837852, "loss": 0.0149, "grad_norm": 4.979900360107422, "learning_rate": 4.963636363636364e-06, "num_tokens": 3735532.0, "completions/mean_length": 72.75, "completions/min_length": 69.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.75, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9954915046691895, "rewards/meter/std": 0.002510088263079524, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9954915046691895, "rewards/total_composite/std": 0.002510088263079524, "reward": 0.9954915046691895, "reward_std": 0.002510099671781063, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06602619588375092, "sampling/sampling_logp_difference/max": 1.1701393127441406, "sampling/importance_sampling_ratio/min": 0.31032371520996094, "sampling/importance_sampling_ratio/mean": 1.0033316612243652, "sampling/importance_sampling_ratio/max": 1.7458137273788452, "entropy": 0.49573158100247383, "clip_ratio/low_mean": 0.010420631850138307, "clip_ratio/low_min": 0.010420631850138307, "clip_ratio/high_mean": 0.03229976072907448, "clip_ratio/high_max": 0.03229976072907448, "clip_ratio/region_mean": 0.042720392579212785, "reward_total_mean": 0.9954915046691895, "reward_meter_mean": 0.9954915046691895, "reward_meter_std": 0.002510088263079524, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9954915046691895, "reward_total_composite_std": 0.002510088263079524} {"timestamp_utc": "2026-04-12T00:37:34Z", "mode": "train", "global_step": 1664, "epoch": 0.06683536169016348, "loss": -0.0026, "grad_norm": 3.2529075145721436, "learning_rate": 4.9606060606060605e-06, "num_tokens": 3737762.0, "completions/mean_length": 109.75, "completions/min_length": 102.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 109.75, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.9983501434326172, "rewards/meter/std": 0.000316560355713591, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9983501434326172, "rewards/total_composite/std": 0.000316560355713591, "reward": 0.9983501434326172, "reward_std": 0.0003165603266097605, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05091632902622223, "sampling/sampling_logp_difference/max": 1.1261930465698242, "sampling/importance_sampling_ratio/min": 0.32426539063453674, "sampling/importance_sampling_ratio/mean": 1.0035275220870972, "sampling/importance_sampling_ratio/max": 1.7752270698547363, "entropy": 0.3692820519208908, "clip_ratio/low_mean": 0.004454409005120397, "clip_ratio/low_min": 0.004454409005120397, "clip_ratio/high_mean": 0.028268360998481512, "clip_ratio/high_max": 0.028268360998481512, "clip_ratio/region_mean": 0.03272277000360191, "reward_total_mean": 0.9983501434326172, "reward_meter_mean": 0.9983501434326172, "reward_meter_std": 0.000316560355713591, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9983501434326172, "reward_total_composite_std": 0.000316560355713591} {"timestamp_utc": "2026-04-12T00:37:42Z", "mode": "train", "global_step": 1665, "epoch": 0.06687552717194843, "loss": 0.0217, "grad_norm": 2.9181759357452393, "learning_rate": 4.957575757575758e-06, "num_tokens": 3742119.0, "completions/mean_length": 340.625, "completions/min_length": 320.0, "completions/max_length": 366.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 340.625, "completions/min_terminated_length": 320.0, "completions/max_terminated_length": 366.0, "rewards/meter/mean": 0.9840777516365051, "rewards/meter/std": 0.037384092807769775, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.7647058963775635, "rewards/repeat_penalty/std": 0.10428297519683838, "rewards/total_composite/mean": 0.5716930627822876, "rewards/total_composite/std": 0.2410053312778473, "reward": 0.5716930627822876, "reward_std": 0.2410053312778473, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05633535981178284, "sampling/sampling_logp_difference/max": 10.131498336791992, "sampling/importance_sampling_ratio/min": 3.980578185291961e-05, "sampling/importance_sampling_ratio/mean": 1.0077635049819946, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3913237787783146, "clip_ratio/low_mean": 0.004957506898790598, "clip_ratio/low_min": 0.004957506898790598, "clip_ratio/high_mean": 0.029565565433586016, "clip_ratio/high_max": 0.029565565433586016, "clip_ratio/region_mean": 0.034523072332376614, "reward_total_mean": 0.5716930627822876, "reward_meter_mean": 0.9840777516365051, "reward_meter_std": 0.037384092807769775, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.7647058963775635, "reward_repeat_penalty_std": 0.10428297519683838, "reward_total_composite_mean": 0.5716930627822876, "reward_total_composite_std": 0.2410053312778473} {"timestamp_utc": "2026-04-12T00:37:50Z", "mode": "train", "global_step": 1666, "epoch": 0.06691569265373339, "loss": -0.0177, "grad_norm": 1.908144474029541, "learning_rate": 4.954545454545455e-06, "num_tokens": 3746080.0, "completions/mean_length": 302.125, "completions/min_length": 285.0, "completions/max_length": 314.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 302.125, "completions/min_terminated_length": 285.0, "completions/max_terminated_length": 314.0, "rewards/meter/mean": 0.9341138005256653, "rewards/meter/std": 0.05671977251768112, "rewards/count_adherence/mean": 0.8181818127632141, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.75, "rewards/repeat_penalty/std": 0.08169000595808029, "rewards/total_composite/mean": 0.5741910934448242, "rewards/total_composite/std": 0.07986098527908325, "reward": 0.5741910934448242, "reward_std": 0.07986099272966385, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03923000767827034, "sampling/sampling_logp_difference/max": 2.483020067214966, "sampling/importance_sampling_ratio/min": 0.08349069207906723, "sampling/importance_sampling_ratio/mean": 1.0056406259536743, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2841501086950302, "clip_ratio/low_mean": 0.01092962856637314, "clip_ratio/low_min": 0.01092962856637314, "clip_ratio/high_mean": 0.01824606256559491, "clip_ratio/high_max": 0.01824606256559491, "clip_ratio/region_mean": 0.02917569113196805, "reward_total_mean": 0.5741910934448242, "reward_meter_mean": 0.9341138005256653, "reward_meter_std": 0.05671977251768112, "reward_count_adherence_mean": 0.8181818127632141, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.75, "reward_repeat_penalty_std": 0.08169000595808029, "reward_total_composite_mean": 0.5741910934448242, "reward_total_composite_std": 0.07986098527908325} {"timestamp_utc": "2026-04-12T00:37:55Z", "mode": "train", "global_step": 1667, "epoch": 0.06695585813551834, "loss": 0.0056, "grad_norm": 2.4326391220092773, "learning_rate": 4.951515151515152e-06, "num_tokens": 3748310.0, "completions/mean_length": 101.75, "completions/min_length": 101.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.75, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.9970037937164307, "rewards/meter/std": 0.0013122691307216883, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7976030111312866, "rewards/total_composite/std": 0.0010498042684048414, "reward": 0.7976030111312866, "reward_std": 0.0010498074116185308, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010477752424776554, "sampling/sampling_logp_difference/max": 1.087967872619629, "sampling/importance_sampling_ratio/min": 0.33690041303634644, "sampling/importance_sampling_ratio/mean": 1.0012123584747314, "sampling/importance_sampling_ratio/max": 1.46913480758667, "entropy": 0.08447205042466521, "clip_ratio/low_mean": 0.0024390824837610126, "clip_ratio/low_min": 0.0024390824837610126, "clip_ratio/high_mean": 0.004938361467793584, "clip_ratio/high_max": 0.004938361467793584, "clip_ratio/region_mean": 0.007377443951554596, "reward_total_mean": 0.7976030111312866, "reward_meter_mean": 0.9970037937164307, "reward_meter_std": 0.0013122691307216883, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7976030111312866, "reward_total_composite_std": 0.0010498042684048414} {"timestamp_utc": "2026-04-12T00:38:00Z", "mode": "train", "global_step": 1668, "epoch": 0.0669960236173033, "loss": -0.0005, "grad_norm": 0.08446517586708069, "learning_rate": 4.9484848484848495e-06, "num_tokens": 3750844.0, "completions/mean_length": 133.75, "completions/min_length": 133.0, "completions/max_length": 134.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 133.75, "completions/min_terminated_length": 133.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.9979548454284668, "rewards/meter/std": 2.2615500711253844e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7142857313156128, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7128248810768127, "rewards/total_composite/std": 1.6168985894182697e-05, "reward": 0.7128248810768127, "reward_std": 1.6179135855054483e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004637876525521278, "sampling/sampling_logp_difference/max": 1.0087156295776367, "sampling/importance_sampling_ratio/min": 0.36468705534935, "sampling/importance_sampling_ratio/mean": 1.001842975616455, "sampling/importance_sampling_ratio/max": 1.2750511169433594, "entropy": 0.027194248279556632, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7128248810768127, "reward_meter_mean": 0.9979548454284668, "reward_meter_std": 2.2615500711253844e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7142857313156128, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7128248810768127, "reward_total_composite_std": 1.6168985894182697e-05} {"timestamp_utc": "2026-04-12T00:38:05Z", "mode": "train", "global_step": 1669, "epoch": 0.06703618909908825, "loss": -0.0009, "grad_norm": 0.0685642808675766, "learning_rate": 4.945454545454546e-06, "num_tokens": 3752596.0, "completions/mean_length": 68.0, "completions/min_length": 67.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9978952407836914, "rewards/meter/std": 1.1280608305241913e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978952407836914, "rewards/total_composite/std": 1.1280608305241913e-05, "reward": 0.9978952407836914, "reward_std": 1.128060739574721e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004578685853630304, "sampling/sampling_logp_difference/max": 0.41336488723754883, "sampling/importance_sampling_ratio/min": 0.6614209413528442, "sampling/importance_sampling_ratio/mean": 1.0022841691970825, "sampling/importance_sampling_ratio/max": 1.0789562463760376, "entropy": 0.03746737586334348, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0018115942366421223, "clip_ratio/high_max": 0.0018115942366421223, "clip_ratio/region_mean": 0.0018115942366421223, "reward_total_mean": 0.9978952407836914, "reward_meter_mean": 0.9978952407836914, "reward_meter_std": 1.1280608305241913e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9978952407836914, "reward_total_composite_std": 1.1280608305241913e-05} {"timestamp_utc": "2026-04-12T00:38:09Z", "mode": "train", "global_step": 1670, "epoch": 0.0670763545808732, "loss": 0.0003, "grad_norm": 3.68241024017334, "learning_rate": 4.942424242424243e-06, "num_tokens": 3754401.0, "completions/mean_length": 63.625, "completions/min_length": 62.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.625, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.986209511756897, "rewards/meter/std": 0.004416565876454115, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.986209511756897, "rewards/total_composite/std": 0.004416565876454115, "reward": 0.986209511756897, "reward_std": 0.004416569136083126, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.029973594471812248, "sampling/sampling_logp_difference/max": 1.5556812286376953, "sampling/importance_sampling_ratio/min": 0.2110455483198166, "sampling/importance_sampling_ratio/mean": 1.000848650932312, "sampling/importance_sampling_ratio/max": 1.363709807395935, "entropy": 0.1983039677143097, "clip_ratio/low_mean": 0.004032257944345474, "clip_ratio/low_min": 0.004032257944345474, "clip_ratio/high_mean": 0.01764108322095126, "clip_ratio/high_max": 0.01764108322095126, "clip_ratio/region_mean": 0.021673341165296733, "reward_total_mean": 0.986209511756897, "reward_meter_mean": 0.986209511756897, "reward_meter_std": 0.004416565876454115, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.986209511756897, "reward_total_composite_std": 0.004416565876454115} {"timestamp_utc": "2026-04-12T00:38:14Z", "mode": "train", "global_step": 1671, "epoch": 0.06711652006265816, "loss": -0.0002, "grad_norm": 1.1145938634872437, "learning_rate": 4.93939393939394e-06, "num_tokens": 3756527.0, "completions/mean_length": 100.75, "completions/min_length": 100.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.75, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.9979295134544373, "rewards/meter/std": 2.0676665371865965e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8250000476837158, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8232920169830322, "rewards/total_composite/std": 0.07056734710931778, "reward": 0.8232920169830322, "reward_std": 0.07056733220815659, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0027057162951678038, "sampling/sampling_logp_difference/max": 0.21077418327331543, "sampling/importance_sampling_ratio/min": 0.809956967830658, "sampling/importance_sampling_ratio/mean": 1.0012245178222656, "sampling/importance_sampling_ratio/max": 1.0940825939178467, "entropy": 0.022657749010249972, "clip_ratio/low_mean": 0.0012499999720603228, "clip_ratio/low_min": 0.0012499999720603228, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0012499999720603228, "reward_total_mean": 0.8232920169830322, "reward_meter_mean": 0.9979295134544373, "reward_meter_std": 2.0676665371865965e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8250000476837158, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.8232920169830322, "reward_total_composite_std": 0.07056734710931778} {"timestamp_utc": "2026-04-12T00:38:19Z", "mode": "train", "global_step": 1672, "epoch": 0.06715668554444311, "loss": -0.0112, "grad_norm": 4.178384304046631, "learning_rate": 4.936363636363637e-06, "num_tokens": 3758078.0, "completions/mean_length": 32.875, "completions/min_length": 32.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.875, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9963486194610596, "rewards/meter/std": 0.006714941468089819, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9963486194610596, "rewards/total_composite/std": 0.006714941468089819, "reward": 0.9963486194610596, "reward_std": 0.006714944262057543, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003993004094809294, "sampling/sampling_logp_difference/max": 0.5090913772583008, "sampling/importance_sampling_ratio/min": 0.6010414361953735, "sampling/importance_sampling_ratio/mean": 1.000542163848877, "sampling/importance_sampling_ratio/max": 1.1358792781829834, "entropy": 0.02497886586934328, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9963486194610596, "reward_meter_mean": 0.9963486194610596, "reward_meter_std": 0.006714941468089819, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9963486194610596, "reward_total_composite_std": 0.006714941468089819} {"timestamp_utc": "2026-04-12T00:38:24Z", "mode": "train", "global_step": 1673, "epoch": 0.06719685102622806, "loss": 0.0159, "grad_norm": 5.359276294708252, "learning_rate": 4.933333333333334e-06, "num_tokens": 3760034.0, "completions/mean_length": 73.5, "completions/min_length": 68.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.5, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.995732307434082, "rewards/meter/std": 0.00316053768619895, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.995732307434082, "rewards/total_composite/std": 0.00316053768619895, "reward": 0.995732307434082, "reward_std": 0.0031605414114892483, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05547928065061569, "sampling/sampling_logp_difference/max": 1.5614430904388428, "sampling/importance_sampling_ratio/min": 0.2098330408334732, "sampling/importance_sampling_ratio/mean": 1.0138006210327148, "sampling/importance_sampling_ratio/max": 1.7035573720932007, "entropy": 0.47178683429956436, "clip_ratio/low_mean": 0.008515364723280072, "clip_ratio/low_min": 0.008515364723280072, "clip_ratio/high_mean": 0.0351892322069034, "clip_ratio/high_max": 0.0351892322069034, "clip_ratio/region_mean": 0.04370459693018347, "reward_total_mean": 0.995732307434082, "reward_meter_mean": 0.995732307434082, "reward_meter_std": 0.00316053768619895, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.995732307434082, "reward_total_composite_std": 0.00316053768619895} {"timestamp_utc": "2026-04-12T00:38:29Z", "mode": "train", "global_step": 1674, "epoch": 0.06723701650801302, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.9303030303030305e-06, "num_tokens": 3761666.0, "completions/mean_length": 58.0, "completions/min_length": 58.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.993520200252533, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.993520200252533, "rewards/total_composite/std": 0.0, "reward": 0.993520200252533, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0007125565898604691, "sampling/sampling_logp_difference/max": 0.018451236188411713, "sampling/importance_sampling_ratio/min": 0.9973061680793762, "sampling/importance_sampling_ratio/mean": 1.0006948709487915, "sampling/importance_sampling_ratio/max": 1.0186225175857544, "entropy": 0.005282787780743092, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.993520200252533, "reward_meter_mean": 0.993520200252533, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.993520200252533, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T00:38:33Z", "mode": "train", "global_step": 1675, "epoch": 0.06727718198979797, "loss": 0.0014, "grad_norm": 0.6250240802764893, "learning_rate": 4.927272727272728e-06, "num_tokens": 3763698.0, "completions/mean_length": 86.0, "completions/min_length": 86.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.0, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.9938453435897827, "rewards/meter/std": 0.0002583391033113003, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7950762510299683, "rewards/total_composite/std": 0.00020666708587668836, "reward": 0.7950762510299683, "reward_std": 0.00020667610806412995, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0035054178442806005, "sampling/sampling_logp_difference/max": 1.0392670631408691, "sampling/importance_sampling_ratio/min": 0.9702885746955872, "sampling/importance_sampling_ratio/mean": 1.0030723810195923, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.017064974061213434, "clip_ratio/low_mean": 0.0014534883666783571, "clip_ratio/low_min": 0.0014534883666783571, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0014534883666783571, "reward_total_mean": 0.7950762510299683, "reward_meter_mean": 0.9938453435897827, "reward_meter_std": 0.0002583391033113003, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7950762510299683, "reward_total_composite_std": 0.00020666708587668836} {"timestamp_utc": "2026-04-12T00:38:39Z", "mode": "train", "global_step": 1676, "epoch": 0.06731734747158293, "loss": 0.0003, "grad_norm": 1.5943388938903809, "learning_rate": 4.924242424242425e-06, "num_tokens": 3765698.0, "completions/mean_length": 101.0, "completions/min_length": 100.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.0, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.9978747963905334, "rewards/meter/std": 0.0001633708452573046, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.8731323480606079, "rewards/total_composite/std": 0.10320712625980377, "reward": 0.8731323480606079, "reward_std": 0.10320711135864258, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008153069764375687, "sampling/sampling_logp_difference/max": 1.007601261138916, "sampling/importance_sampling_ratio/min": 0.3650937080383301, "sampling/importance_sampling_ratio/mean": 1.0001577138900757, "sampling/importance_sampling_ratio/max": 1.161500334739685, "entropy": 0.04971472639590502, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.00492622796446085, "clip_ratio/high_max": 0.00492622796446085, "clip_ratio/region_mean": 0.00492622796446085, "reward_total_mean": 0.8731323480606079, "reward_meter_mean": 0.9978747963905334, "reward_meter_std": 0.0001633708452573046, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.8731323480606079, "reward_total_composite_std": 0.10320712625980377} {"timestamp_utc": "2026-04-12T00:38:43Z", "mode": "train", "global_step": 1677, "epoch": 0.06735751295336788, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.9212121212121214e-06, "num_tokens": 3767450.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9982472658157349, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9982472658157349, "rewards/total_composite/std": 0.0, "reward": 0.9982472658157349, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0014719502069056034, "sampling/sampling_logp_difference/max": 0.2051541954278946, "sampling/importance_sampling_ratio/min": 0.8145217299461365, "sampling/importance_sampling_ratio/mean": 1.0002412796020508, "sampling/importance_sampling_ratio/max": 1.0495675802230835, "entropy": 0.00934242527000606, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9982472658157349, "reward_meter_mean": 0.9982472658157349, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9982472658157349, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T00:38:48Z", "mode": "train", "global_step": 1678, "epoch": 0.06739767843515283, "loss": 0.01, "grad_norm": 1.9516417980194092, "learning_rate": 4.918181818181819e-06, "num_tokens": 3769712.0, "completions/mean_length": 99.75, "completions/min_length": 98.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.75, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.9527222514152527, "rewards/meter/std": 0.0017279081512242556, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7857142686843872, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.748537540435791, "rewards/total_composite/std": 0.0724104791879654, "reward": 0.748537540435791, "reward_std": 0.0724104791879654, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01071135699748993, "sampling/sampling_logp_difference/max": 1.6311063766479492, "sampling/importance_sampling_ratio/min": 0.19571292400360107, "sampling/importance_sampling_ratio/mean": 0.999224066734314, "sampling/importance_sampling_ratio/max": 1.2951877117156982, "entropy": 0.04197088209912181, "clip_ratio/low_mean": 0.0012376237427815795, "clip_ratio/low_min": 0.0012376237427815795, "clip_ratio/high_mean": 0.0037499999161809683, "clip_ratio/high_max": 0.0037499999161809683, "clip_ratio/region_mean": 0.004987623658962548, "reward_total_mean": 0.748537540435791, "reward_meter_mean": 0.9527222514152527, "reward_meter_std": 0.0017279081512242556, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7857142686843872, "reward_repeat_penalty_std": 0.07636035233736038, "reward_total_composite_mean": 0.748537540435791, "reward_total_composite_std": 0.0724104791879654} {"timestamp_utc": "2026-04-12T00:38:55Z", "mode": "train", "global_step": 1679, "epoch": 0.06743784391693779, "loss": -0.006, "grad_norm": 1.1392688751220703, "learning_rate": 4.915151515151516e-06, "num_tokens": 3772903.0, "completions/mean_length": 216.875, "completions/min_length": 214.0, "completions/max_length": 224.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 216.875, "completions/min_terminated_length": 214.0, "completions/max_terminated_length": 224.0, "rewards/meter/mean": 0.9947628974914551, "rewards/meter/std": 0.0007815666613169014, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6442307829856873, "rewards/repeat_penalty/std": 0.039811473339796066, "rewards/total_composite/mean": 0.5340501070022583, "rewards/total_composite/std": 0.03305770829319954, "reward": 0.5340501070022583, "reward_std": 0.03305771201848984, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009834333322942257, "sampling/sampling_logp_difference/max": 1.1391210556030273, "sampling/importance_sampling_ratio/min": 0.32010024785995483, "sampling/importance_sampling_ratio/mean": 1.0005079507827759, "sampling/importance_sampling_ratio/max": 1.3776015043258667, "entropy": 0.052406568080186844, "clip_ratio/low_mean": 0.0011655074777081609, "clip_ratio/low_min": 0.0011655074777081609, "clip_ratio/high_mean": 0.003399575361981988, "clip_ratio/high_max": 0.003399575361981988, "clip_ratio/region_mean": 0.004565082839690149, "reward_total_mean": 0.5340501070022583, "reward_meter_mean": 0.9947628974914551, "reward_meter_std": 0.0007815666613169014, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6442307829856873, "reward_repeat_penalty_std": 0.039811473339796066, "reward_total_composite_mean": 0.5340501070022583, "reward_total_composite_std": 0.03305770829319954} {"timestamp_utc": "2026-04-12T00:39:04Z", "mode": "train", "global_step": 1680, "epoch": 0.06747800939872274, "loss": -0.0168, "grad_norm": 2.291543960571289, "learning_rate": 4.912121212121212e-06, "num_tokens": 3777950.0, "completions/mean_length": 401.875, "completions/min_length": 373.0, "completions/max_length": 455.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 401.875, "completions/min_terminated_length": 373.0, "completions/max_terminated_length": 455.0, "rewards/meter/mean": 0.9955718517303467, "rewards/meter/std": 0.004510289058089256, "rewards/count_adherence/mean": 0.7232142686843872, "rewards/count_adherence/std": 0.02525380253791809, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8195801973342896, "rewards/repeat_penalty/std": 0.1243370845913887, "rewards/total_composite/mean": 0.5898604393005371, "rewards/total_composite/std": 0.08871299028396606, "reward": 0.5898604393005371, "reward_std": 0.08871299028396606, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0480833500623703, "sampling/sampling_logp_difference/max": 1.780477523803711, "sampling/importance_sampling_ratio/min": 0.17153088748455048, "sampling/importance_sampling_ratio/mean": 1.0117672681808472, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3997117578983307, "clip_ratio/low_mean": 0.010193796595558524, "clip_ratio/low_min": 0.010193796595558524, "clip_ratio/high_mean": 0.0216394632589072, "clip_ratio/high_max": 0.0216394632589072, "clip_ratio/region_mean": 0.03183325985446572, "reward_total_mean": 0.5898604393005371, "reward_meter_mean": 0.9955718517303467, "reward_meter_std": 0.004510289058089256, "reward_count_adherence_mean": 0.7232142686843872, "reward_count_adherence_std": 0.02525380253791809, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8195801973342896, "reward_repeat_penalty_std": 0.1243370845913887, "reward_total_composite_mean": 0.5898604393005371, "reward_total_composite_std": 0.08871299028396606} {"timestamp_utc": "2026-04-12T00:39:08Z", "mode": "train", "global_step": 1681, "epoch": 0.0675181748805077, "loss": -0.0134, "grad_norm": 4.46686315536499, "learning_rate": 4.90909090909091e-06, "num_tokens": 3779464.0, "completions/mean_length": 30.25, "completions/min_length": 29.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.25, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.9945183992385864, "rewards/meter/std": 0.000778909248765558, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9945183992385864, "rewards/total_composite/std": 0.000778909248765558, "reward": 0.9945183992385864, "reward_std": 0.0007789283408783376, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0202788058668375, "sampling/sampling_logp_difference/max": 0.6706576347351074, "sampling/importance_sampling_ratio/min": 0.5113722085952759, "sampling/importance_sampling_ratio/mean": 0.9981259703636169, "sampling/importance_sampling_ratio/max": 1.514021873474121, "entropy": 0.1191922826692462, "clip_ratio/low_mean": 0.008477011695504189, "clip_ratio/low_min": 0.008477011695504189, "clip_ratio/high_mean": 0.02016128972172737, "clip_ratio/high_max": 0.02016128972172737, "clip_ratio/region_mean": 0.02863830141723156, "reward_total_mean": 0.9945183992385864, "reward_meter_mean": 0.9945183992385864, "reward_meter_std": 0.000778909248765558, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9945183992385864, "reward_total_composite_std": 0.000778909248765558} {"timestamp_utc": "2026-04-12T00:39:13Z", "mode": "train", "global_step": 1682, "epoch": 0.06755834036229265, "loss": -0.0204, "grad_norm": 7.099659442901611, "learning_rate": 4.906060606060606e-06, "num_tokens": 3781340.0, "completions/mean_length": 72.5, "completions/min_length": 67.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.5, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9961867928504944, "rewards/meter/std": 0.0030412215273827314, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9961867928504944, "rewards/total_composite/std": 0.0030412215273827314, "reward": 0.9961867928504944, "reward_std": 0.003041210351511836, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06393711268901825, "sampling/sampling_logp_difference/max": 1.4125175476074219, "sampling/importance_sampling_ratio/min": 0.24352942407131195, "sampling/importance_sampling_ratio/mean": 1.0079271793365479, "sampling/importance_sampling_ratio/max": 1.853277325630188, "entropy": 0.46719325333833694, "clip_ratio/low_mean": 0.007042253389954567, "clip_ratio/low_min": 0.007042253389954567, "clip_ratio/high_mean": 0.0440298137255013, "clip_ratio/high_max": 0.0440298137255013, "clip_ratio/region_mean": 0.051072067115455866, "reward_total_mean": 0.9961867928504944, "reward_meter_mean": 0.9961867928504944, "reward_meter_std": 0.0030412215273827314, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9961867928504944, "reward_total_composite_std": 0.0030412215273827314} {"timestamp_utc": "2026-04-12T00:39:17Z", "mode": "train", "global_step": 1683, "epoch": 0.0675985058440776, "loss": -0.0016, "grad_norm": 0.45891278982162476, "learning_rate": 4.903030303030303e-06, "num_tokens": 3783091.0, "completions/mean_length": 63.875, "completions/min_length": 63.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.875, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9982389211654663, "rewards/meter/std": 2.360223516006954e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9982389211654663, "rewards/total_composite/std": 2.360223516006954e-05, "reward": 0.9982389211654663, "reward_std": 2.360223516006954e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004287322983145714, "sampling/sampling_logp_difference/max": 0.9284515380859375, "sampling/importance_sampling_ratio/min": 0.3951651453971863, "sampling/importance_sampling_ratio/mean": 1.000044345855713, "sampling/importance_sampling_ratio/max": 1.0655800104141235, "entropy": 0.01849041902460158, "clip_ratio/low_mean": 0.0019841270986944437, "clip_ratio/low_min": 0.0019841270986944437, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0019841270986944437, "reward_total_mean": 0.9982389211654663, "reward_meter_mean": 0.9982389211654663, "reward_meter_std": 2.360223516006954e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9982389211654663, "reward_total_composite_std": 2.360223516006954e-05} {"timestamp_utc": "2026-04-12T00:39:23Z", "mode": "train", "global_step": 1684, "epoch": 0.06763867132586256, "loss": 0.0107, "grad_norm": 3.898653507232666, "learning_rate": 4.9000000000000005e-06, "num_tokens": 3785743.0, "completions/mean_length": 159.5, "completions/min_length": 148.0, "completions/max_length": 180.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 159.5, "completions/min_terminated_length": 148.0, "completions/max_terminated_length": 180.0, "rewards/meter/mean": 0.9947781562805176, "rewards/meter/std": 0.007610421162098646, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9769483804702759, "rewards/total_composite/std": 0.049505408853292465, "reward": 0.9769483804702759, "reward_std": 0.049505408853292465, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.059815503656864166, "sampling/sampling_logp_difference/max": 1.4108057022094727, "sampling/importance_sampling_ratio/min": 0.24394665658473969, "sampling/importance_sampling_ratio/mean": 1.0116267204284668, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5180013440549374, "clip_ratio/low_mean": 0.007966701406985521, "clip_ratio/low_min": 0.007966701406985521, "clip_ratio/high_mean": 0.024855362717062235, "clip_ratio/high_max": 0.024855362717062235, "clip_ratio/region_mean": 0.032822064124047756, "reward_total_mean": 0.9769483804702759, "reward_meter_mean": 0.9947781562805176, "reward_meter_std": 0.007610421162098646, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.9769483804702759, "reward_total_composite_std": 0.049505408853292465} {"timestamp_utc": "2026-04-12T00:39:28Z", "mode": "train", "global_step": 1685, "epoch": 0.06767883680764751, "loss": -0.0028, "grad_norm": 2.095104932785034, "learning_rate": 4.896969696969697e-06, "num_tokens": 3787673.0, "completions/mean_length": 64.25, "completions/min_length": 64.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.25, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9981728792190552, "rewards/meter/std": 0.0002840534143615514, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981728792190552, "rewards/total_composite/std": 0.0002840534143615514, "reward": 0.9981728792190552, "reward_std": 0.00028403810574673116, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011698653921484947, "sampling/sampling_logp_difference/max": 1.6340885162353516, "sampling/importance_sampling_ratio/min": 0.1951301395893097, "sampling/importance_sampling_ratio/mean": 0.9983230829238892, "sampling/importance_sampling_ratio/max": 1.5041022300720215, "entropy": 0.04154683533124626, "clip_ratio/low_mean": 0.005859375, "clip_ratio/low_min": 0.005859375, "clip_ratio/high_mean": 0.0037878789007663727, "clip_ratio/high_max": 0.0037878789007663727, "clip_ratio/region_mean": 0.009647253900766373, "reward_total_mean": 0.9981728792190552, "reward_meter_mean": 0.9981728792190552, "reward_meter_std": 0.0002840534143615514, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981728792190552, "reward_total_composite_std": 0.0002840534143615514} {"timestamp_utc": "2026-04-12T00:39:32Z", "mode": "train", "global_step": 1686, "epoch": 0.06771900228943246, "loss": -0.0182, "grad_norm": 4.037059783935547, "learning_rate": 4.893939393939394e-06, "num_tokens": 3789623.0, "completions/mean_length": 71.75, "completions/min_length": 68.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.75, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.997489333152771, "rewards/meter/std": 0.0010296773398295045, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.997489333152771, "rewards/total_composite/std": 0.0010296773398295045, "reward": 0.997489333152771, "reward_std": 0.0010296726832166314, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04423650726675987, "sampling/sampling_logp_difference/max": 1.0650873184204102, "sampling/importance_sampling_ratio/min": 0.3446977436542511, "sampling/importance_sampling_ratio/mean": 1.0071872472763062, "sampling/importance_sampling_ratio/max": 1.5631378889083862, "entropy": 0.4106294997036457, "clip_ratio/low_mean": 0.017920455895364285, "clip_ratio/low_min": 0.017920455895364285, "clip_ratio/high_mean": 0.02395429043099284, "clip_ratio/high_max": 0.02395429043099284, "clip_ratio/region_mean": 0.041874746326357126, "reward_total_mean": 0.997489333152771, "reward_meter_mean": 0.997489333152771, "reward_meter_std": 0.0010296773398295045, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.997489333152771, "reward_total_composite_std": 0.0010296773398295045} {"timestamp_utc": "2026-04-12T00:39:36Z", "mode": "train", "global_step": 1687, "epoch": 0.06775916777121742, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.8909090909090914e-06, "num_tokens": 3791023.0, "completions/mean_length": 35.0, "completions/min_length": 35.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9977574944496155, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977574944496155, "rewards/total_composite/std": 0.0, "reward": 0.9977574944496155, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0031034064013510942, "sampling/sampling_logp_difference/max": 0.04160819947719574, "sampling/importance_sampling_ratio/min": 0.9605222940444946, "sampling/importance_sampling_ratio/mean": 1.0023516416549683, "sampling/importance_sampling_ratio/max": 1.0424859523773193, "entropy": 0.02776301186531782, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9977574944496155, "reward_meter_mean": 0.9977574944496155, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9977574944496155, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T00:39:40Z", "mode": "train", "global_step": 1688, "epoch": 0.06779933325300237, "loss": 0.0003, "grad_norm": 4.740636348724365, "learning_rate": 4.887878787878788e-06, "num_tokens": 3792659.0, "completions/mean_length": 35.5, "completions/min_length": 35.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.5, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9980167150497437, "rewards/meter/std": 0.00040671354508958757, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980167150497437, "rewards/total_composite/std": 0.00040671354508958757, "reward": 0.9980167150497437, "reward_std": 0.0004067234694957733, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021727347746491432, "sampling/sampling_logp_difference/max": 0.9272527694702148, "sampling/importance_sampling_ratio/min": 0.3956391513347626, "sampling/importance_sampling_ratio/mean": 0.9994746446609497, "sampling/importance_sampling_ratio/max": 1.5504966974258423, "entropy": 0.12218342442065477, "clip_ratio/low_mean": 0.0070436508394777775, "clip_ratio/low_min": 0.0070436508394777775, "clip_ratio/high_mean": 0.0035714285913854837, "clip_ratio/high_max": 0.0035714285913854837, "clip_ratio/region_mean": 0.010615079430863261, "reward_total_mean": 0.9980167150497437, "reward_meter_mean": 0.9980167150497437, "reward_meter_std": 0.00040671354508958757, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9980167150497437, "reward_total_composite_std": 0.00040671354508958757} {"timestamp_utc": "2026-04-12T00:39:45Z", "mode": "train", "global_step": 1689, "epoch": 0.06783949873478733, "loss": 0.0024, "grad_norm": 0.21065974235534668, "learning_rate": 4.884848484848485e-06, "num_tokens": 3794387.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.998286783695221, "rewards/meter/std": 0.00011179452121723443, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998286783695221, "rewards/total_composite/std": 0.00011179452121723443, "reward": 0.998286783695221, "reward_std": 0.00011178548447787762, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0037898647133260965, "sampling/sampling_logp_difference/max": 1.470503807067871, "sampling/importance_sampling_ratio/min": 0.22980967164039612, "sampling/importance_sampling_ratio/mean": 0.9991194009780884, "sampling/importance_sampling_ratio/max": 1.028564214706421, "entropy": 0.008520788454916328, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.998286783695221, "reward_meter_mean": 0.998286783695221, "reward_meter_std": 0.00011179452121723443, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.998286783695221, "reward_total_composite_std": 0.00011179452121723443} {"timestamp_utc": "2026-04-12T00:39:49Z", "mode": "train", "global_step": 1690, "epoch": 0.06787966421657228, "loss": 0.0367, "grad_norm": 7.852858066558838, "learning_rate": 4.881818181818182e-06, "num_tokens": 3795915.0, "completions/mean_length": 36.0, "completions/min_length": 34.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.934417724609375, "rewards/meter/std": 0.06754057109355927, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.934417724609375, "rewards/total_composite/std": 0.06754057109355927, "reward": 0.934417724609375, "reward_std": 0.06754057854413986, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05185196176171303, "sampling/sampling_logp_difference/max": 1.1850357055664062, "sampling/importance_sampling_ratio/min": 0.3057352602481842, "sampling/importance_sampling_ratio/mean": 1.0148731470108032, "sampling/importance_sampling_ratio/max": 1.7753301858901978, "entropy": 0.3176651671528816, "clip_ratio/low_mean": 0.003289473708719015, "clip_ratio/low_min": 0.003289473708719015, "clip_ratio/high_mean": 0.04497848777100444, "clip_ratio/high_max": 0.04497848777100444, "clip_ratio/region_mean": 0.048267961479723454, "reward_total_mean": 0.934417724609375, "reward_meter_mean": 0.934417724609375, "reward_meter_std": 0.06754057109355927, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.934417724609375, "reward_total_composite_std": 0.06754057109355927} {"timestamp_utc": "2026-04-12T00:39:54Z", "mode": "train", "global_step": 1691, "epoch": 0.06791982969835723, "loss": -0.0057, "grad_norm": 3.046787738800049, "learning_rate": 4.878787878787879e-06, "num_tokens": 3798615.0, "completions/mean_length": 140.5, "completions/min_length": 136.0, "completions/max_length": 146.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 140.5, "completions/min_terminated_length": 136.0, "completions/max_terminated_length": 146.0, "rewards/meter/mean": 0.9408473372459412, "rewards/meter/std": 0.04903604835271835, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.09155284613370895, "rewards/total_composite/mean": 0.824957013130188, "rewards/total_composite/std": 0.11124026030302048, "reward": 0.824957013130188, "reward_std": 0.11124025285243988, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04019925743341446, "sampling/sampling_logp_difference/max": 1.1345267295837402, "sampling/importance_sampling_ratio/min": 0.32157430052757263, "sampling/importance_sampling_ratio/mean": 1.012131690979004, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.28699430637061596, "clip_ratio/low_mean": 0.014275046763941646, "clip_ratio/low_min": 0.014275046763941646, "clip_ratio/high_mean": 0.025541604205500335, "clip_ratio/high_max": 0.025541604205500335, "clip_ratio/region_mean": 0.03981665096944198, "reward_total_mean": 0.824957013130188, "reward_meter_mean": 0.9408473372459412, "reward_meter_std": 0.04903604835271835, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.09155284613370895, "reward_total_composite_mean": 0.824957013130188, "reward_total_composite_std": 0.11124026030302048} {"timestamp_utc": "2026-04-12T00:40:00Z", "mode": "train", "global_step": 1692, "epoch": 0.06795999518014219, "loss": -0.0153, "grad_norm": 2.5923614501953125, "learning_rate": 4.875757575757576e-06, "num_tokens": 3801374.0, "completions/mean_length": 148.875, "completions/min_length": 142.0, "completions/max_length": 156.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 148.875, "completions/min_terminated_length": 142.0, "completions/max_terminated_length": 156.0, "rewards/meter/mean": 0.9931907653808594, "rewards/meter/std": 0.0014422648819163442, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6305556297302246, "rewards/repeat_penalty/std": 0.0602024681866169, "rewards/total_composite/mean": 0.6263105869293213, "rewards/total_composite/std": 0.060353342443704605, "reward": 0.6263105869293213, "reward_std": 0.06035333871841431, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009429208934307098, "sampling/sampling_logp_difference/max": 1.152205228805542, "sampling/importance_sampling_ratio/min": 0.3159392774105072, "sampling/importance_sampling_ratio/mean": 1.001092791557312, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03733880212530494, "clip_ratio/low_mean": 0.005935311957728118, "clip_ratio/low_min": 0.005935311957728118, "clip_ratio/high_mean": 0.006003972492180765, "clip_ratio/high_max": 0.006003972492180765, "clip_ratio/region_mean": 0.011939284449908882, "reward_total_mean": 0.6263105869293213, "reward_meter_mean": 0.9931907653808594, "reward_meter_std": 0.0014422648819163442, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6305556297302246, "reward_repeat_penalty_std": 0.0602024681866169, "reward_total_composite_mean": 0.6263105869293213, "reward_total_composite_std": 0.060353342443704605} {"timestamp_utc": "2026-04-12T00:40:04Z", "mode": "train", "global_step": 1693, "epoch": 0.06800016066192714, "loss": 0.013, "grad_norm": 5.695197105407715, "learning_rate": 4.872727272727273e-06, "num_tokens": 3803064.0, "completions/mean_length": 67.25, "completions/min_length": 62.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.25, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.7432924509048462, "rewards/meter/std": 0.2945479154586792, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7432924509048462, "rewards/total_composite/std": 0.2945479154586792, "reward": 0.7432924509048462, "reward_std": 0.2945479154586792, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05113214999437332, "sampling/sampling_logp_difference/max": 1.1394152641296387, "sampling/importance_sampling_ratio/min": 0.3200061023235321, "sampling/importance_sampling_ratio/mean": 1.0079994201660156, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40255092084407806, "clip_ratio/low_mean": 0.009146409342065454, "clip_ratio/low_min": 0.009146409342065454, "clip_ratio/high_mean": 0.02981708408333361, "clip_ratio/high_max": 0.02981708408333361, "clip_ratio/region_mean": 0.038963493425399065, "reward_total_mean": 0.7432924509048462, "reward_meter_mean": 0.7432924509048462, "reward_meter_std": 0.2945479154586792, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7432924509048462, "reward_total_composite_std": 0.2945479154586792} {"timestamp_utc": "2026-04-12T00:40:10Z", "mode": "train", "global_step": 1694, "epoch": 0.0680403261437121, "loss": 0.0364, "grad_norm": 2.8474440574645996, "learning_rate": 4.8696969696969705e-06, "num_tokens": 3805883.0, "completions/mean_length": 169.375, "completions/min_length": 158.0, "completions/max_length": 182.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 169.375, "completions/min_terminated_length": 158.0, "completions/max_terminated_length": 182.0, "rewards/meter/mean": 0.6835033893585205, "rewards/meter/std": 0.2869633436203003, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9041666388511658, "rewards/repeat_penalty/std": 0.07100624591112137, "rewards/total_composite/mean": 0.6048145294189453, "rewards/total_composite/std": 0.23514944314956665, "reward": 0.6048145294189453, "reward_std": 0.23514942824840546, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03621907904744148, "sampling/sampling_logp_difference/max": 0.9100509285926819, "sampling/importance_sampling_ratio/min": 0.40250372886657715, "sampling/importance_sampling_ratio/mean": 1.00909423828125, "sampling/importance_sampling_ratio/max": 1.8559550046920776, "entropy": 0.3172535989433527, "clip_ratio/low_mean": 0.010541909374296665, "clip_ratio/low_min": 0.010541909374296665, "clip_ratio/high_mean": 0.020258112344890833, "clip_ratio/high_max": 0.020258112344890833, "clip_ratio/region_mean": 0.030800021719187498, "reward_total_mean": 0.6048145294189453, "reward_meter_mean": 0.6835033893585205, "reward_meter_std": 0.2869633436203003, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9041666388511658, "reward_repeat_penalty_std": 0.07100624591112137, "reward_total_composite_mean": 0.6048145294189453, "reward_total_composite_std": 0.23514944314956665} {"timestamp_utc": "2026-04-12T00:40:15Z", "mode": "train", "global_step": 1695, "epoch": 0.06808049162549705, "loss": -0.0183, "grad_norm": 2.8666982650756836, "learning_rate": 4.866666666666667e-06, "num_tokens": 3807608.0, "completions/mean_length": 60.625, "completions/min_length": 58.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.625, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.8901118040084839, "rewards/meter/std": 0.2941354215145111, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8901118040084839, "rewards/total_composite/std": 0.2941354215145111, "reward": 0.8901118040084839, "reward_std": 0.2941353917121887, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02997700497508049, "sampling/sampling_logp_difference/max": 1.047898292541504, "sampling/importance_sampling_ratio/min": 0.35067400336265564, "sampling/importance_sampling_ratio/mean": 0.9950342178344727, "sampling/importance_sampling_ratio/max": 1.2877650260925293, "entropy": 0.17649692203849554, "clip_ratio/low_mean": 0.006465517450124025, "clip_ratio/low_min": 0.006465517450124025, "clip_ratio/high_mean": 0.03078524977900088, "clip_ratio/high_max": 0.03078524977900088, "clip_ratio/region_mean": 0.037250767229124904, "reward_total_mean": 0.8901118040084839, "reward_meter_mean": 0.8901118040084839, "reward_meter_std": 0.2941354215145111, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8901118040084839, "reward_total_composite_std": 0.2941354215145111} {"timestamp_utc": "2026-04-12T00:40:20Z", "mode": "train", "global_step": 1696, "epoch": 0.068120657107282, "loss": -0.0017, "grad_norm": 3.8953700065612793, "learning_rate": 4.863636363636364e-06, "num_tokens": 3809772.0, "completions/mean_length": 106.5, "completions/min_length": 103.0, "completions/max_length": 110.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.5, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 110.0, "rewards/meter/mean": 0.9039969444274902, "rewards/meter/std": 0.09514570236206055, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9039969444274902, "rewards/total_composite/std": 0.09514570236206055, "reward": 0.9039969444274902, "reward_std": 0.09514573216438293, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04417286068201065, "sampling/sampling_logp_difference/max": 1.5998225212097168, "sampling/importance_sampling_ratio/min": 0.20193234086036682, "sampling/importance_sampling_ratio/mean": 0.9994289875030518, "sampling/importance_sampling_ratio/max": 1.748896598815918, "entropy": 0.3774713296443224, "clip_ratio/low_mean": 0.010750595480203629, "clip_ratio/low_min": 0.010750595480203629, "clip_ratio/high_mean": 0.03136626980267465, "clip_ratio/high_max": 0.03136626980267465, "clip_ratio/region_mean": 0.04211686528287828, "reward_total_mean": 0.9039969444274902, "reward_meter_mean": 0.9039969444274902, "reward_meter_std": 0.09514570236206055, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9039969444274902, "reward_total_composite_std": 0.09514570236206055} {"timestamp_utc": "2026-04-12T00:40:24Z", "mode": "train", "global_step": 1697, "epoch": 0.06816082258906696, "loss": 0.0098, "grad_norm": 12.74048137664795, "learning_rate": 4.8606060606060615e-06, "num_tokens": 3811557.0, "completions/mean_length": 69.125, "completions/min_length": 64.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.125, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.7374060153961182, "rewards/meter/std": 0.2923416793346405, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7374060153961182, "rewards/total_composite/std": 0.2923416793346405, "reward": 0.7374060153961182, "reward_std": 0.2923416793346405, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.051393892616033554, "sampling/sampling_logp_difference/max": 0.840245246887207, "sampling/importance_sampling_ratio/min": 0.43160465359687805, "sampling/importance_sampling_ratio/mean": 1.0157667398452759, "sampling/importance_sampling_ratio/max": 1.9715321063995361, "entropy": 0.3504845257848501, "clip_ratio/low_mean": 0.011194029822945595, "clip_ratio/low_min": 0.011194029822945595, "clip_ratio/high_mean": 0.026841548271477222, "clip_ratio/high_max": 0.026841548271477222, "clip_ratio/region_mean": 0.03803557809442282, "reward_total_mean": 0.7374060153961182, "reward_meter_mean": 0.7374060153961182, "reward_meter_std": 0.2923416793346405, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7374060153961182, "reward_total_composite_std": 0.2923416793346405} {"timestamp_utc": "2026-04-12T00:40:34Z", "mode": "train", "global_step": 1698, "epoch": 0.06820098807085191, "loss": 0.0122, "grad_norm": 2.547250747680664, "learning_rate": 4.857575757575758e-06, "num_tokens": 3817138.0, "completions/mean_length": 456.625, "completions/min_length": 442.0, "completions/max_length": 476.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 456.625, "completions/min_terminated_length": 442.0, "completions/max_terminated_length": 476.0, "rewards/meter/mean": 0.9678418636322021, "rewards/meter/std": 0.03973681107163429, "rewards/count_adherence/mean": 0.7678571343421936, "rewards/count_adherence/std": 0.033064987510442734, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9239448308944702, "rewards/repeat_penalty/std": 0.04794442281126976, "rewards/total_composite/mean": 0.6861064434051514, "rewards/total_composite/std": 0.04557538405060768, "reward": 0.6861064434051514, "reward_std": 0.04557539150118828, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.055158257484436035, "sampling/sampling_logp_difference/max": 2.418172836303711, "sampling/importance_sampling_ratio/min": 0.08908424526453018, "sampling/importance_sampling_ratio/mean": 1.0073450803756714, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4843989312648773, "clip_ratio/low_mean": 0.01929042232222855, "clip_ratio/low_min": 0.01929042232222855, "clip_ratio/high_mean": 0.023344212910160422, "clip_ratio/high_max": 0.023344212910160422, "clip_ratio/region_mean": 0.04263463523238897, "reward_total_mean": 0.6861064434051514, "reward_meter_mean": 0.9678418636322021, "reward_meter_std": 0.03973681107163429, "reward_count_adherence_mean": 0.7678571343421936, "reward_count_adherence_std": 0.033064987510442734, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9239448308944702, "reward_repeat_penalty_std": 0.04794442281126976, "reward_total_composite_mean": 0.6861064434051514, "reward_total_composite_std": 0.04557538405060768} {"timestamp_utc": "2026-04-12T00:40:38Z", "mode": "train", "global_step": 1699, "epoch": 0.06824115355263687, "loss": -0.0066, "grad_norm": 3.725172758102417, "learning_rate": 4.854545454545455e-06, "num_tokens": 3818997.0, "completions/mean_length": 72.375, "completions/min_length": 70.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.375, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9986300468444824, "rewards/meter/std": 0.0003518488083500415, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9986300468444824, "rewards/total_composite/std": 0.0003518488083500415, "reward": 0.9986300468444824, "reward_std": 0.00035184345324523747, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03500673174858093, "sampling/sampling_logp_difference/max": 1.481827974319458, "sampling/importance_sampling_ratio/min": 0.22722196578979492, "sampling/importance_sampling_ratio/mean": 1.0039602518081665, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23103010654449463, "clip_ratio/low_mean": 0.010468951310031116, "clip_ratio/low_min": 0.010468951310031116, "clip_ratio/high_mean": 0.01543669670354575, "clip_ratio/high_max": 0.01543669670354575, "clip_ratio/region_mean": 0.025905648013576865, "reward_total_mean": 0.9986300468444824, "reward_meter_mean": 0.9986300468444824, "reward_meter_std": 0.0003518488083500415, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9986300468444824, "reward_total_composite_std": 0.0003518488083500415} {"timestamp_utc": "2026-04-12T00:40:44Z", "mode": "train", "global_step": 1700, "epoch": 0.06828131903442182, "loss": 0.0025, "grad_norm": 1.9409257173538208, "learning_rate": 4.851515151515152e-06, "num_tokens": 3822000.0, "completions/mean_length": 171.375, "completions/min_length": 170.0, "completions/max_length": 173.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 171.375, "completions/min_terminated_length": 170.0, "completions/max_terminated_length": 173.0, "rewards/meter/mean": 0.9947851896286011, "rewards/meter/std": 0.0019038140308111906, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6875, "rewards/repeat_penalty/std": 0.035355325788259506, "rewards/total_composite/mean": 0.6838877201080322, "rewards/total_composite/std": 0.034654706716537476, "reward": 0.6838877201080322, "reward_std": 0.034654706716537476, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014711027964949608, "sampling/sampling_logp_difference/max": 6.82924747467041, "sampling/importance_sampling_ratio/min": 0.0010816717986017466, "sampling/importance_sampling_ratio/mean": 0.9983268976211548, "sampling/importance_sampling_ratio/max": 1.4856215715408325, "entropy": 0.04710668884217739, "clip_ratio/low_mean": 0.0007267441833391786, "clip_ratio/low_min": 0.0007267441833391786, "clip_ratio/high_mean": 0.008750928391236812, "clip_ratio/high_max": 0.008750928391236812, "clip_ratio/region_mean": 0.00947767257457599, "reward_total_mean": 0.6838877201080322, "reward_meter_mean": 0.9947851896286011, "reward_meter_std": 0.0019038140308111906, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6875, "reward_repeat_penalty_std": 0.035355325788259506, "reward_total_composite_mean": 0.6838877201080322, "reward_total_composite_std": 0.034654706716537476} {"timestamp_utc": "2026-04-12T00:41:57Z", "mode": "eval", "global_step": 1700, "epoch": 0.06828131903442182, "eval_loss": NaN, "eval_runtime": 72.6468, "eval_samples_per_second": 1.432, "eval_steps_per_second": 0.179, "eval_num_tokens": 3822000.0, "eval_completions/mean_length": 213.16346153846155, "eval_completions/min_length": 64.15384615384616, "eval_completions/max_length": 386.46153846153845, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 213.16346153846155, "eval_completions/min_terminated_length": 64.15384615384616, "eval_completions/max_terminated_length": 386.46153846153845, "eval_rewards/meter/mean": 0.7185193575345553, "eval_rewards/meter/std": 0.4064074754714966, "eval_rewards/count_adherence/mean": 0.9180220273824838, "eval_rewards/count_adherence/std": 0.10367171552318794, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/repeat_penalty/mean": 0.7850257295828599, "eval_rewards/repeat_penalty/std": 0.15948278972735772, "eval_rewards/total_composite/mean": 0.5148563522558945, "eval_rewards/total_composite/std": 0.3435696088350736, "eval_reward": 0.5148563522558945, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.020774336352657814, "eval_sampling/sampling_logp_difference/max": 1.5070369427020733, "eval_sampling/importance_sampling_ratio/min": 0.3100252260382359, "eval_sampling/importance_sampling_ratio/mean": 1.0051932701697717, "eval_sampling/importance_sampling_ratio/max": 1.4555618029374342, "eval_entropy": 0.20546403641884142, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5148563522558945, "eval_reward_meter_mean": 0.7185193575345553, "eval_reward_meter_std": 0.4064074754714966, "eval_reward_count_adherence_mean": 0.9180220273824838, "eval_reward_count_adherence_std": 0.10367171552318794, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_repeat_penalty_mean": 0.7850257295828599, "eval_reward_repeat_penalty_std": 0.15948278972735772, "eval_reward_total_composite_mean": 0.5148563522558945, "eval_reward_total_composite_std": 0.3435696088350736} {"timestamp_utc": "2026-04-12T00:42:06Z", "mode": "train", "global_step": 1701, "epoch": 0.06832148451620677, "loss": 0.0094, "grad_norm": 2.8516652584075928, "learning_rate": 4.848484848484849e-06, "num_tokens": 3825370.0, "completions/mean_length": 247.25, "completions/min_length": 228.0, "completions/max_length": 266.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 247.25, "completions/min_terminated_length": 228.0, "completions/max_terminated_length": 266.0, "rewards/meter/mean": 0.8367831110954285, "rewards/meter/std": 0.21646961569786072, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7960164546966553, "rewards/repeat_penalty/std": 0.12943987548351288, "rewards/total_composite/mean": 0.6609899997711182, "rewards/total_composite/std": 0.1882331371307373, "reward": 0.6609899997711182, "reward_std": 0.1882331222295761, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04899187758564949, "sampling/sampling_logp_difference/max": 11.93899917602539, "sampling/importance_sampling_ratio/min": 6.5306817305099685e-06, "sampling/importance_sampling_ratio/mean": 1.0024902820587158, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2732624653726816, "clip_ratio/low_mean": 0.008105989079922438, "clip_ratio/low_min": 0.008105989079922438, "clip_ratio/high_mean": 0.02072087489068508, "clip_ratio/high_max": 0.02072087489068508, "clip_ratio/region_mean": 0.02882686397060752, "reward_total_mean": 0.6609899997711182, "reward_meter_mean": 0.8367831110954285, "reward_meter_std": 0.21646961569786072, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7960164546966553, "reward_repeat_penalty_std": 0.12943987548351288, "reward_total_composite_mean": 0.6609899997711182, "reward_total_composite_std": 0.1882331371307373} {"timestamp_utc": "2026-04-12T00:42:11Z", "mode": "train", "global_step": 1702, "epoch": 0.06836164999799173, "loss": 0.013, "grad_norm": 3.22273325920105, "learning_rate": 4.845454545454546e-06, "num_tokens": 3827574.0, "completions/mean_length": 108.5, "completions/min_length": 102.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.5, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.9983426332473755, "rewards/meter/std": 0.000583991059102118, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.8985103964805603, "rewards/total_composite/std": 0.10674867779016495, "reward": 0.8985103964805603, "reward_std": 0.10674867779016495, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03188902884721756, "sampling/sampling_logp_difference/max": 1.428539752960205, "sampling/importance_sampling_ratio/min": 0.2396586388349533, "sampling/importance_sampling_ratio/mean": 1.0079683065414429, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.24685455113649368, "clip_ratio/low_mean": 0.011560323182493448, "clip_ratio/low_min": 0.011560323182493448, "clip_ratio/high_mean": 0.012898168293759227, "clip_ratio/high_max": 0.012898168293759227, "clip_ratio/region_mean": 0.024458491476252675, "reward_total_mean": 0.8985103964805603, "reward_meter_mean": 0.9983426332473755, "reward_meter_std": 0.000583991059102118, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_total_composite_mean": 0.8985103964805603, "reward_total_composite_std": 0.10674867779016495} {"timestamp_utc": "2026-04-12T00:42:19Z", "mode": "train", "global_step": 1703, "epoch": 0.06840181547977668, "loss": 0.0086, "grad_norm": 1.9336153268814087, "learning_rate": 4.842424242424243e-06, "num_tokens": 3832143.0, "completions/mean_length": 328.125, "completions/min_length": 316.0, "completions/max_length": 352.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 328.125, "completions/min_terminated_length": 316.0, "completions/max_terminated_length": 352.0, "rewards/meter/mean": 0.9899157285690308, "rewards/meter/std": 0.008645469322800636, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.03928370773792267, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9504464268684387, "rewards/repeat_penalty/std": 0.030713941901922226, "rewards/total_composite/mean": 0.8237563371658325, "rewards/total_composite/std": 0.054851170629262924, "reward": 0.8237563371658325, "reward_std": 0.054851170629262924, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05240955203771591, "sampling/sampling_logp_difference/max": 2.064100742340088, "sampling/importance_sampling_ratio/min": 0.12693238258361816, "sampling/importance_sampling_ratio/mean": 1.0071643590927124, "sampling/importance_sampling_ratio/max": 1.909447431564331, "entropy": 0.4287557154893875, "clip_ratio/low_mean": 0.017736590933054686, "clip_ratio/low_min": 0.017736590933054686, "clip_ratio/high_mean": 0.019598021870478988, "clip_ratio/high_max": 0.019598021870478988, "clip_ratio/region_mean": 0.03733461280353367, "reward_total_mean": 0.8237563371658325, "reward_meter_mean": 0.9899157285690308, "reward_meter_std": 0.008645469322800636, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.03928370773792267, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9504464268684387, "reward_repeat_penalty_std": 0.030713941901922226, "reward_total_composite_mean": 0.8237563371658325, "reward_total_composite_std": 0.054851170629262924} {"timestamp_utc": "2026-04-12T00:42:25Z", "mode": "train", "global_step": 1704, "epoch": 0.06844198096156164, "loss": 0.0231, "grad_norm": 4.468354225158691, "learning_rate": 4.83939393939394e-06, "num_tokens": 3834840.0, "completions/mean_length": 143.125, "completions/min_length": 140.0, "completions/max_length": 151.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 143.125, "completions/min_terminated_length": 140.0, "completions/max_terminated_length": 151.0, "rewards/meter/mean": 0.9751874804496765, "rewards/meter/std": 0.05766330659389496, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.75, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7313905954360962, "rewards/total_composite/std": 0.04324747994542122, "reward": 0.7313905954360962, "reward_std": 0.04324747994542122, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013407315127551556, "sampling/sampling_logp_difference/max": 1.5817131996154785, "sampling/importance_sampling_ratio/min": 0.20562252402305603, "sampling/importance_sampling_ratio/mean": 1.0032126903533936, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.05452722427435219, "clip_ratio/low_mean": 0.002483443822711706, "clip_ratio/low_min": 0.002483443822711706, "clip_ratio/high_mean": 0.008767852385062724, "clip_ratio/high_max": 0.008767852385062724, "clip_ratio/region_mean": 0.01125129620777443, "reward_total_mean": 0.7313905954360962, "reward_meter_mean": 0.9751874804496765, "reward_meter_std": 0.05766330659389496, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.75, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7313905954360962, "reward_total_composite_std": 0.04324747994542122} {"timestamp_utc": "2026-04-12T00:42:29Z", "mode": "train", "global_step": 1705, "epoch": 0.06848214644334659, "loss": 0.0492, "grad_norm": 9.636635780334473, "learning_rate": 4.836363636363637e-06, "num_tokens": 3836711.0, "completions/mean_length": 75.875, "completions/min_length": 71.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.875, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9914894104003906, "rewards/meter/std": 0.009747824631631374, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9914894104003906, "rewards/total_composite/std": 0.009747824631631374, "reward": 0.9914894104003906, "reward_std": 0.009747819975018501, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07412047684192657, "sampling/sampling_logp_difference/max": 1.237552523612976, "sampling/importance_sampling_ratio/min": 0.290093332529068, "sampling/importance_sampling_ratio/mean": 1.0037219524383545, "sampling/importance_sampling_ratio/max": 1.6591025590896606, "entropy": 0.66333282366395, "clip_ratio/low_mean": 0.02736378228291869, "clip_ratio/low_min": 0.02736378228291869, "clip_ratio/high_mean": 0.033178919460624456, "clip_ratio/high_max": 0.033178919460624456, "clip_ratio/region_mean": 0.06054270174354315, "reward_total_mean": 0.9914894104003906, "reward_meter_mean": 0.9914894104003906, "reward_meter_std": 0.009747824631631374, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9914894104003906, "reward_total_composite_std": 0.009747824631631374} {"timestamp_utc": "2026-04-12T00:42:37Z", "mode": "train", "global_step": 1706, "epoch": 0.06852231192513154, "loss": -0.016, "grad_norm": 1.8596045970916748, "learning_rate": 4.833333333333333e-06, "num_tokens": 3840705.0, "completions/mean_length": 302.25, "completions/min_length": 279.0, "completions/max_length": 322.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 302.25, "completions/min_terminated_length": 279.0, "completions/max_terminated_length": 322.0, "rewards/meter/mean": 0.998089075088501, "rewards/meter/std": 0.0013965349644422531, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.604687511920929, "rewards/repeat_penalty/std": 0.15109451115131378, "rewards/total_composite/mean": 0.6033692955970764, "rewards/total_composite/std": 0.14974084496498108, "reward": 0.6033692955970764, "reward_std": 0.1497408151626587, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022071924060583115, "sampling/sampling_logp_difference/max": 7.479898929595947, "sampling/importance_sampling_ratio/min": 0.0005643144831992686, "sampling/importance_sampling_ratio/mean": 1.0016717910766602, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.13442990742623806, "clip_ratio/low_mean": 0.007105766620952636, "clip_ratio/low_min": 0.007105766620952636, "clip_ratio/high_mean": 0.0056348692160099745, "clip_ratio/high_max": 0.0056348692160099745, "clip_ratio/region_mean": 0.01274063583696261, "reward_total_mean": 0.6033692955970764, "reward_meter_mean": 0.998089075088501, "reward_meter_std": 0.0013965349644422531, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.604687511920929, "reward_repeat_penalty_std": 0.15109451115131378, "reward_total_composite_mean": 0.6033692955970764, "reward_total_composite_std": 0.14974084496498108} {"timestamp_utc": "2026-04-12T00:42:41Z", "mode": "train", "global_step": 1707, "epoch": 0.0685624774069165, "loss": 0.0035, "grad_norm": 4.459820747375488, "learning_rate": 4.830303030303031e-06, "num_tokens": 3842546.0, "completions/mean_length": 61.125, "completions/min_length": 60.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.125, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9614025354385376, "rewards/meter/std": 0.046098750084638596, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9614025354385376, "rewards/total_composite/std": 0.046098750084638596, "reward": 0.9614025354385376, "reward_std": 0.046098742634058, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0242039505392313, "sampling/sampling_logp_difference/max": 0.7878568172454834, "sampling/importance_sampling_ratio/min": 0.5145033597946167, "sampling/importance_sampling_ratio/mean": 1.005699872970581, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1731706503778696, "clip_ratio/low_mean": 0.004166666883975267, "clip_ratio/low_min": 0.004166666883975267, "clip_ratio/high_mean": 0.014213158516213298, "clip_ratio/high_max": 0.014213158516213298, "clip_ratio/region_mean": 0.018379825400188565, "reward_total_mean": 0.9614025354385376, "reward_meter_mean": 0.9614025354385376, "reward_meter_std": 0.046098750084638596, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9614025354385376, "reward_total_composite_std": 0.046098750084638596} {"timestamp_utc": "2026-04-12T00:42:46Z", "mode": "train", "global_step": 1708, "epoch": 0.06860264288870145, "loss": -0.0028, "grad_norm": 4.370044231414795, "learning_rate": 4.827272727272728e-06, "num_tokens": 3844472.0, "completions/mean_length": 75.75, "completions/min_length": 72.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.75, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9969611167907715, "rewards/meter/std": 0.0019538968335837126, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9969611167907715, "rewards/total_composite/std": 0.0019538968335837126, "reward": 0.9969611167907715, "reward_std": 0.001953905913978815, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.048442739993333817, "sampling/sampling_logp_difference/max": 1.1050148010253906, "sampling/importance_sampling_ratio/min": 0.33120596408843994, "sampling/importance_sampling_ratio/mean": 1.0110034942626953, "sampling/importance_sampling_ratio/max": 1.6920305490493774, "entropy": 0.360767001286149, "clip_ratio/low_mean": 0.011602062964811921, "clip_ratio/low_min": 0.011602062964811921, "clip_ratio/high_mean": 0.019748708698898554, "clip_ratio/high_max": 0.019748708698898554, "clip_ratio/region_mean": 0.031350771663710475, "reward_total_mean": 0.9969611167907715, "reward_meter_mean": 0.9969611167907715, "reward_meter_std": 0.0019538968335837126, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9969611167907715, "reward_total_composite_std": 0.0019538968335837126} {"timestamp_utc": "2026-04-12T00:42:51Z", "mode": "train", "global_step": 1709, "epoch": 0.0686428083704864, "loss": 0.0066, "grad_norm": 7.5510640144348145, "learning_rate": 4.824242424242424e-06, "num_tokens": 3846403.0, "completions/mean_length": 79.375, "completions/min_length": 71.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.375, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.9834517240524292, "rewards/meter/std": 0.016320018097758293, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9834517240524292, "rewards/total_composite/std": 0.016320018097758293, "reward": 0.9834517240524292, "reward_std": 0.016320008784532547, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06894521415233612, "sampling/sampling_logp_difference/max": 1.456099510192871, "sampling/importance_sampling_ratio/min": 0.2331438809633255, "sampling/importance_sampling_ratio/mean": 1.019029974937439, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.663462370634079, "clip_ratio/low_mean": 0.029091242235153913, "clip_ratio/low_min": 0.029091242235153913, "clip_ratio/high_mean": 0.029255983070470393, "clip_ratio/high_max": 0.029255983070470393, "clip_ratio/region_mean": 0.058347225305624306, "reward_total_mean": 0.9834517240524292, "reward_meter_mean": 0.9834517240524292, "reward_meter_std": 0.016320018097758293, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9834517240524292, "reward_total_composite_std": 0.016320018097758293} {"timestamp_utc": "2026-04-12T00:42:56Z", "mode": "train", "global_step": 1710, "epoch": 0.06868297385227136, "loss": 0.0194, "grad_norm": 2.7458198070526123, "learning_rate": 4.8212121212121215e-06, "num_tokens": 3848628.0, "completions/mean_length": 105.125, "completions/min_length": 102.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.125, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.9941129684448242, "rewards/meter/std": 0.0036593328695744276, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9941129684448242, "rewards/total_composite/std": 0.0036593328695744276, "reward": 0.9941129684448242, "reward_std": 0.0036593314725905657, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020809238776564598, "sampling/sampling_logp_difference/max": 1.1916732788085938, "sampling/importance_sampling_ratio/min": 0.3037126362323761, "sampling/importance_sampling_ratio/mean": 1.0030418634414673, "sampling/importance_sampling_ratio/max": 1.5832405090332031, "entropy": 0.17343183141201735, "clip_ratio/low_mean": 0.01390780950896442, "clip_ratio/low_min": 0.01390780950896442, "clip_ratio/high_mean": 0.008437702199444175, "clip_ratio/high_max": 0.008437702199444175, "clip_ratio/region_mean": 0.022345511708408594, "reward_total_mean": 0.9941129684448242, "reward_meter_mean": 0.9941129684448242, "reward_meter_std": 0.0036593328695744276, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9941129684448242, "reward_total_composite_std": 0.0036593328695744276} {"timestamp_utc": "2026-04-12T00:43:00Z", "mode": "train", "global_step": 1711, "epoch": 0.06872313933405631, "loss": 0.0035, "grad_norm": 6.050206661224365, "learning_rate": 4.818181818181819e-06, "num_tokens": 3850396.0, "completions/mean_length": 62.0, "completions/min_length": 60.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.0, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.992821455001831, "rewards/meter/std": 0.0031417824793606997, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.992821455001831, "rewards/total_composite/std": 0.0031417824793606997, "reward": 0.992821455001831, "reward_std": 0.00314177293330431, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028909733518958092, "sampling/sampling_logp_difference/max": 0.8604459762573242, "sampling/importance_sampling_ratio/min": 0.4229734241962433, "sampling/importance_sampling_ratio/mean": 1.0075486898422241, "sampling/importance_sampling_ratio/max": 1.8755117654800415, "entropy": 0.21017488650977612, "clip_ratio/low_mean": 0.010147797176614404, "clip_ratio/low_min": 0.010147797176614404, "clip_ratio/high_mean": 0.010149948066100478, "clip_ratio/high_max": 0.010149948066100478, "clip_ratio/region_mean": 0.020297745242714882, "reward_total_mean": 0.992821455001831, "reward_meter_mean": 0.992821455001831, "reward_meter_std": 0.0031417824793606997, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.992821455001831, "reward_total_composite_std": 0.0031417824793606997} {"timestamp_utc": "2026-04-12T00:43:04Z", "mode": "train", "global_step": 1712, "epoch": 0.06876330481584127, "loss": 0.0252, "grad_norm": 9.417886734008789, "learning_rate": 4.815151515151515e-06, "num_tokens": 3851781.0, "completions/mean_length": 31.125, "completions/min_length": 30.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.125, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.99363112449646, "rewards/meter/std": 0.0030946105252951384, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.99363112449646, "rewards/total_composite/std": 0.0030946105252951384, "reward": 0.99363112449646, "reward_std": 0.0030946091283112764, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01712997443974018, "sampling/sampling_logp_difference/max": 0.6930625438690186, "sampling/importance_sampling_ratio/min": 0.5000423192977905, "sampling/importance_sampling_ratio/mean": 1.0098103284835815, "sampling/importance_sampling_ratio/max": 1.6804507970809937, "entropy": 0.138029710855335, "clip_ratio/low_mean": 0.011742424685508013, "clip_ratio/low_min": 0.011742424685508013, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.011742424685508013, "reward_total_mean": 0.99363112449646, "reward_meter_mean": 0.99363112449646, "reward_meter_std": 0.0030946105252951384, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.99363112449646, "reward_total_composite_std": 0.0030946105252951384} {"timestamp_utc": "2026-04-12T00:43:09Z", "mode": "train", "global_step": 1713, "epoch": 0.06880347029762622, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.8121212121212125e-06, "num_tokens": 3853917.0, "completions/mean_length": 86.0, "completions/min_length": 86.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.0, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.9939366579055786, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7951493263244629, "rewards/total_composite/std": 0.0, "reward": 0.7951493263244629, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00075820047641173, "sampling/sampling_logp_difference/max": 0.02473144233226776, "sampling/importance_sampling_ratio/min": 0.975571870803833, "sampling/importance_sampling_ratio/mean": 1.0006271600723267, "sampling/importance_sampling_ratio/max": 1.0226460695266724, "entropy": 0.007212614000309259, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7951493263244629, "reward_meter_mean": 0.9939366579055786, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7951493263244629, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T00:43:13Z", "mode": "train", "global_step": 1714, "epoch": 0.06884363577941117, "loss": 0.0002, "grad_norm": 0.25375768542289734, "learning_rate": 4.80909090909091e-06, "num_tokens": 3855829.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9982520937919617, "rewards/meter/std": 1.3676654816663358e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9982520937919617, "rewards/total_composite/std": 1.3676654816663358e-05, "reward": 0.9982520937919617, "reward_std": 1.3667625353264157e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0014053680934011936, "sampling/sampling_logp_difference/max": 0.2431938648223877, "sampling/importance_sampling_ratio/min": 0.7841194868087769, "sampling/importance_sampling_ratio/mean": 0.9998658895492554, "sampling/importance_sampling_ratio/max": 1.0173074007034302, "entropy": 0.007174990780185908, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9982520937919617, "reward_meter_mean": 0.9982520937919617, "reward_meter_std": 1.3676654816663358e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9982520937919617, "reward_total_composite_std": 1.3676654816663358e-05} {"timestamp_utc": "2026-04-12T00:43:18Z", "mode": "train", "global_step": 1715, "epoch": 0.06888380126119613, "loss": 0.0006, "grad_norm": 0.6869344115257263, "learning_rate": 4.806060606060606e-06, "num_tokens": 3857509.0, "completions/mean_length": 50.0, "completions/min_length": 50.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.0, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.9374866485595703, "rewards/meter/std": 3.270595334470272e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9374866485595703, "rewards/total_composite/std": 3.270595334470272e-05, "reward": 0.9374866485595703, "reward_std": 3.270595334470272e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006228248123079538, "sampling/sampling_logp_difference/max": 0.5891187191009521, "sampling/importance_sampling_ratio/min": 0.5548160076141357, "sampling/importance_sampling_ratio/mean": 1.0022356510162354, "sampling/importance_sampling_ratio/max": 1.3050650358200073, "entropy": 0.04075565282255411, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0024999999441206455, "clip_ratio/high_max": 0.0024999999441206455, "clip_ratio/region_mean": 0.0024999999441206455, "reward_total_mean": 0.9374866485595703, "reward_meter_mean": 0.9374866485595703, "reward_meter_std": 3.270595334470272e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9374866485595703, "reward_total_composite_std": 3.270595334470272e-05} {"timestamp_utc": "2026-04-12T00:43:22Z", "mode": "train", "global_step": 1716, "epoch": 0.06892396674298108, "loss": -0.0025, "grad_norm": 1.8621673583984375, "learning_rate": 4.803030303030303e-06, "num_tokens": 3859301.0, "completions/mean_length": 58.0, "completions/min_length": 58.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9931608438491821, "rewards/meter/std": 0.0010164134437218308, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9931608438491821, "rewards/total_composite/std": 0.0010164134437218308, "reward": 0.9931608438491821, "reward_std": 0.001016413327306509, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003477482357993722, "sampling/sampling_logp_difference/max": 1.2498140335083008, "sampling/importance_sampling_ratio/min": 0.2865580916404724, "sampling/importance_sampling_ratio/mean": 0.9991500377655029, "sampling/importance_sampling_ratio/max": 1.0107558965682983, "entropy": 0.007712913560681045, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9931608438491821, "reward_meter_mean": 0.9931608438491821, "reward_meter_std": 0.0010164134437218308, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9931608438491821, "reward_total_composite_std": 0.0010164134437218308} {"timestamp_utc": "2026-04-12T00:43:32Z", "mode": "train", "global_step": 1717, "epoch": 0.06896413222476604, "loss": 0.0087, "grad_norm": 2.0890252590179443, "learning_rate": 4.800000000000001e-06, "num_tokens": 3864531.0, "completions/mean_length": 403.75, "completions/min_length": 377.0, "completions/max_length": 424.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 403.75, "completions/min_terminated_length": 377.0, "completions/max_terminated_length": 424.0, "rewards/meter/mean": 0.9169327616691589, "rewards/meter/std": 0.19917452335357666, "rewards/count_adherence/mean": 0.6583333015441895, "rewards/count_adherence/std": 0.0235702246427536, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.9466373920440674, "rewards/repeat_penalty/std": 0.04092821478843689, "rewards/total_composite/mean": 0.49961453676223755, "rewards/total_composite/std": 0.24736955761909485, "reward": 0.49961453676223755, "reward_std": 0.24736955761909485, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.049914389848709106, "sampling/sampling_logp_difference/max": 1.5311355590820312, "sampling/importance_sampling_ratio/min": 0.2162899225950241, "sampling/importance_sampling_ratio/mean": 1.0099897384643555, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4459982290863991, "clip_ratio/low_mean": 0.007035988033749163, "clip_ratio/low_min": 0.007035988033749163, "clip_ratio/high_mean": 0.033190221060067415, "clip_ratio/high_max": 0.033190221060067415, "clip_ratio/region_mean": 0.04022620909381658, "reward_total_mean": 0.49961453676223755, "reward_meter_mean": 0.9169327616691589, "reward_meter_std": 0.19917452335357666, "reward_count_adherence_mean": 0.6583333015441895, "reward_count_adherence_std": 0.0235702246427536, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.9466373920440674, "reward_repeat_penalty_std": 0.04092821478843689, "reward_total_composite_mean": 0.49961453676223755, "reward_total_composite_std": 0.24736955761909485} {"timestamp_utc": "2026-04-12T00:43:36Z", "mode": "train", "global_step": 1718, "epoch": 0.06900429770655099, "loss": 0.0003, "grad_norm": 0.21243786811828613, "learning_rate": 4.796969696969697e-06, "num_tokens": 3866227.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9982569217681885, "rewards/meter/std": 1.7906948414747603e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9982569217681885, "rewards/total_composite/std": 1.7906948414747603e-05, "reward": 0.9982569217681885, "reward_std": 1.7897747966344468e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00201884051784873, "sampling/sampling_logp_difference/max": 0.3060312271118164, "sampling/importance_sampling_ratio/min": 0.7363636493682861, "sampling/importance_sampling_ratio/mean": 0.9998619556427002, "sampling/importance_sampling_ratio/max": 1.103933572769165, "entropy": 0.009457326959818602, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.001953125, "clip_ratio/high_max": 0.001953125, "clip_ratio/region_mean": 0.001953125, "reward_total_mean": 0.9982569217681885, "reward_meter_mean": 0.9982569217681885, "reward_meter_std": 1.7906948414747603e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9982569217681885, "reward_total_composite_std": 1.7906948414747603e-05} {"timestamp_utc": "2026-04-12T00:43:44Z", "mode": "train", "global_step": 1719, "epoch": 0.06904446318833594, "loss": -0.0103, "grad_norm": 2.303537368774414, "learning_rate": 4.793939393939394e-06, "num_tokens": 3870642.0, "completions/mean_length": 282.875, "completions/min_length": 273.0, "completions/max_length": 292.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 282.875, "completions/min_terminated_length": 273.0, "completions/max_terminated_length": 292.0, "rewards/meter/mean": 0.9829399585723877, "rewards/meter/std": 0.022622600197792053, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.942307710647583, "rewards/repeat_penalty/std": 0.054392825812101364, "rewards/total_composite/mean": 0.9256449937820435, "rewards/total_composite/std": 0.04545941948890686, "reward": 0.9256449937820435, "reward_std": 0.04545941576361656, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06017490476369858, "sampling/sampling_logp_difference/max": 4.142031669616699, "sampling/importance_sampling_ratio/min": 0.015890534967184067, "sampling/importance_sampling_ratio/mean": 1.0142277479171753, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5307335741817951, "clip_ratio/low_mean": 0.021380615420639515, "clip_ratio/low_min": 0.021380615420639515, "clip_ratio/high_mean": 0.02710696868598461, "clip_ratio/high_max": 0.02710696868598461, "clip_ratio/region_mean": 0.048487584106624126, "reward_total_mean": 0.9256449937820435, "reward_meter_mean": 0.9829399585723877, "reward_meter_std": 0.022622600197792053, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.942307710647583, "reward_repeat_penalty_std": 0.054392825812101364, "reward_total_composite_mean": 0.9256449937820435, "reward_total_composite_std": 0.04545941948890686} {"timestamp_utc": "2026-04-12T00:43:53Z", "mode": "train", "global_step": 1720, "epoch": 0.0690846286701209, "loss": -0.0169, "grad_norm": 2.2101805210113525, "learning_rate": 4.790909090909091e-06, "num_tokens": 3875535.0, "completions/mean_length": 400.625, "completions/min_length": 368.0, "completions/max_length": 418.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 400.625, "completions/min_terminated_length": 368.0, "completions/max_terminated_length": 418.0, "rewards/meter/mean": 0.9720591306686401, "rewards/meter/std": 0.04081534594297409, "rewards/count_adherence/mean": 0.6499999761581421, "rewards/count_adherence/std": 0.030860668048262596, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9199131727218628, "rewards/repeat_penalty/std": 0.08860698342323303, "rewards/total_composite/mean": 0.5811381340026855, "rewards/total_composite/std": 0.0678037479519844, "reward": 0.5811381340026855, "reward_std": 0.06780374050140381, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05924157425761223, "sampling/sampling_logp_difference/max": 1.3779230117797852, "sampling/importance_sampling_ratio/min": 0.2521016299724579, "sampling/importance_sampling_ratio/mean": 1.0108215808868408, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5437066555023193, "clip_ratio/low_mean": 0.01499476860044524, "clip_ratio/low_min": 0.01499476860044524, "clip_ratio/high_mean": 0.02708326978608966, "clip_ratio/high_max": 0.02708326978608966, "clip_ratio/region_mean": 0.0420780383865349, "reward_total_mean": 0.5811381340026855, "reward_meter_mean": 0.9720591306686401, "reward_meter_std": 0.04081534594297409, "reward_count_adherence_mean": 0.6499999761581421, "reward_count_adherence_std": 0.030860668048262596, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9199131727218628, "reward_repeat_penalty_std": 0.08860698342323303, "reward_total_composite_mean": 0.5811381340026855, "reward_total_composite_std": 0.0678037479519844} {"timestamp_utc": "2026-04-12T00:43:58Z", "mode": "train", "global_step": 1721, "epoch": 0.06912479415190585, "loss": 0.0135, "grad_norm": 5.097805023193359, "learning_rate": 4.787878787878788e-06, "num_tokens": 3877383.0, "completions/mean_length": 72.0, "completions/min_length": 70.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.0, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9644453525543213, "rewards/meter/std": 0.02713916078209877, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9644453525543213, "rewards/total_composite/std": 0.02713916078209877, "reward": 0.9644453525543213, "reward_std": 0.027139168232679367, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0413968488574028, "sampling/sampling_logp_difference/max": 1.1499383449554443, "sampling/importance_sampling_ratio/min": 0.31665629148483276, "sampling/importance_sampling_ratio/mean": 1.0127274990081787, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3467879444360733, "clip_ratio/low_mean": 0.013723055250011384, "clip_ratio/low_min": 0.013723055250011384, "clip_ratio/high_mean": 0.02464998373761773, "clip_ratio/high_max": 0.02464998373761773, "clip_ratio/region_mean": 0.038373038987629116, "reward_total_mean": 0.9644453525543213, "reward_meter_mean": 0.9644453525543213, "reward_meter_std": 0.02713916078209877, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9644453525543213, "reward_total_composite_std": 0.02713916078209877} {"timestamp_utc": "2026-04-12T00:44:06Z", "mode": "train", "global_step": 1722, "epoch": 0.0691649596336908, "loss": 0.0331, "grad_norm": 2.7361412048339844, "learning_rate": 4.784848484848485e-06, "num_tokens": 3881642.0, "completions/mean_length": 319.375, "completions/min_length": 307.0, "completions/max_length": 340.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 319.375, "completions/min_terminated_length": 307.0, "completions/max_terminated_length": 340.0, "rewards/meter/mean": 0.7120033502578735, "rewards/meter/std": 0.35525479912757874, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.1321374922990799, "rewards/total_composite/mean": 0.6435312032699585, "rewards/total_composite/std": 0.32674068212509155, "reward": 0.6435312032699585, "reward_std": 0.32674068212509155, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06873981654644012, "sampling/sampling_logp_difference/max": 2.4230339527130127, "sampling/importance_sampling_ratio/min": 0.17653696238994598, "sampling/importance_sampling_ratio/mean": 1.011804461479187, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.594971913844347, "clip_ratio/low_mean": 0.012550130486488342, "clip_ratio/low_min": 0.012550130486488342, "clip_ratio/high_mean": 0.04609629465267062, "clip_ratio/high_max": 0.04609629465267062, "clip_ratio/region_mean": 0.058646425139158964, "reward_total_mean": 0.6435312032699585, "reward_meter_mean": 0.7120033502578735, "reward_meter_std": 0.35525479912757874, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.1321374922990799, "reward_total_composite_mean": 0.6435312032699585, "reward_total_composite_std": 0.32674068212509155} {"timestamp_utc": "2026-04-12T00:44:10Z", "mode": "train", "global_step": 1723, "epoch": 0.06920512511547576, "loss": -0.0342, "grad_norm": 6.818909645080566, "learning_rate": 4.7818181818181825e-06, "num_tokens": 3883271.0, "completions/mean_length": 38.625, "completions/min_length": 35.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.625, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.9971029758453369, "rewards/meter/std": 0.0023624140303581953, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9971029758453369, "rewards/total_composite/std": 0.0023624140303581953, "reward": 0.9971029758453369, "reward_std": 0.002362414263188839, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05782073736190796, "sampling/sampling_logp_difference/max": 1.0367670059204102, "sampling/importance_sampling_ratio/min": 0.354599267244339, "sampling/importance_sampling_ratio/mean": 1.012098789215088, "sampling/importance_sampling_ratio/max": 1.597930669784546, "entropy": 0.50562334805727, "clip_ratio/low_mean": 0.017084942432120442, "clip_ratio/low_min": 0.017084942432120442, "clip_ratio/high_mean": 0.028194751124829054, "clip_ratio/high_max": 0.028194751124829054, "clip_ratio/region_mean": 0.045279693556949496, "reward_total_mean": 0.9971029758453369, "reward_meter_mean": 0.9971029758453369, "reward_meter_std": 0.0023624140303581953, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9971029758453369, "reward_total_composite_std": 0.0023624140303581953} {"timestamp_utc": "2026-04-12T00:44:16Z", "mode": "train", "global_step": 1724, "epoch": 0.06924529059726071, "loss": 0.0002, "grad_norm": 0.12045075744390488, "learning_rate": 4.77878787878788e-06, "num_tokens": 3885143.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9982585310935974, "rewards/meter/std": 4.937959602102637e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9982585310935974, "rewards/total_composite/std": 4.937959602102637e-05, "reward": 0.9982585310935974, "reward_std": 4.9387075705453753e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0029545121360570192, "sampling/sampling_logp_difference/max": 0.4872865676879883, "sampling/importance_sampling_ratio/min": 0.6142910122871399, "sampling/importance_sampling_ratio/mean": 1.001138687133789, "sampling/importance_sampling_ratio/max": 1.6037458181381226, "entropy": 0.01320492452941835, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.00390625, "reward_total_mean": 0.9982585310935974, "reward_meter_mean": 0.9982585310935974, "reward_meter_std": 4.937959602102637e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9982585310935974, "reward_total_composite_std": 4.937959602102637e-05} {"timestamp_utc": "2026-04-12T00:44:20Z", "mode": "train", "global_step": 1725, "epoch": 0.06928545607904567, "loss": -0.0117, "grad_norm": 3.2427947521209717, "learning_rate": 4.775757575757576e-06, "num_tokens": 3886827.0, "completions/mean_length": 62.5, "completions/min_length": 59.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.5, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9951238632202148, "rewards/meter/std": 0.0010399873135611415, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9951238632202148, "rewards/total_composite/std": 0.0010399873135611415, "reward": 0.9951238632202148, "reward_std": 0.0010399814927950501, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009362027049064636, "sampling/sampling_logp_difference/max": 0.8125953674316406, "sampling/importance_sampling_ratio/min": 0.4437049925327301, "sampling/importance_sampling_ratio/mean": 1.002244234085083, "sampling/importance_sampling_ratio/max": 1.5123201608657837, "entropy": 0.06342153111472726, "clip_ratio/low_mean": 0.0042372881434857845, "clip_ratio/low_min": 0.0042372881434857845, "clip_ratio/high_mean": 0.0019841270986944437, "clip_ratio/high_max": 0.0019841270986944437, "clip_ratio/region_mean": 0.006221415242180228, "reward_total_mean": 0.9951238632202148, "reward_meter_mean": 0.9951238632202148, "reward_meter_std": 0.0010399873135611415, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9951238632202148, "reward_total_composite_std": 0.0010399873135611415} {"timestamp_utc": "2026-04-12T00:44:26Z", "mode": "train", "global_step": 1726, "epoch": 0.06932562156083062, "loss": 0.0135, "grad_norm": 3.5299017429351807, "learning_rate": 4.772727272727273e-06, "num_tokens": 3889004.0, "completions/mean_length": 130.125, "completions/min_length": 125.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 130.125, "completions/min_terminated_length": 125.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.5966504812240601, "rewards/meter/std": 0.2748834788799286, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8214285373687744, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.48184216022491455, "rewards/total_composite/std": 0.2112974375486374, "reward": 0.48184216022491455, "reward_std": 0.2112974226474762, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03716103360056877, "sampling/sampling_logp_difference/max": 3.358452320098877, "sampling/importance_sampling_ratio/min": 0.034789059311151505, "sampling/importance_sampling_ratio/mean": 1.0085442066192627, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.264286732301116, "clip_ratio/low_mean": 0.011303507490083575, "clip_ratio/low_min": 0.011303507490083575, "clip_ratio/high_mean": 0.009680354851298034, "clip_ratio/high_max": 0.009680354851298034, "clip_ratio/region_mean": 0.02098386234138161, "reward_total_mean": 0.48184216022491455, "reward_meter_mean": 0.5966504812240601, "reward_meter_std": 0.2748834788799286, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8214285373687744, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.48184216022491455, "reward_total_composite_std": 0.2112974375486374} {"timestamp_utc": "2026-04-12T00:44:31Z", "mode": "train", "global_step": 1727, "epoch": 0.06936578704261558, "loss": -0.0028, "grad_norm": 1.0005321502685547, "learning_rate": 4.769696969696971e-06, "num_tokens": 3890696.0, "completions/mean_length": 68.5, "completions/min_length": 67.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.5, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9954667687416077, "rewards/meter/std": 0.004938251804560423, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9954667687416077, "rewards/total_composite/std": 0.004938251804560423, "reward": 0.9954667687416077, "reward_std": 0.004938257858157158, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01310752984136343, "sampling/sampling_logp_difference/max": 0.6610031127929688, "sampling/importance_sampling_ratio/min": 0.5163331627845764, "sampling/importance_sampling_ratio/mean": 1.0034555196762085, "sampling/importance_sampling_ratio/max": 1.4678758382797241, "entropy": 0.11760625522583723, "clip_ratio/low_mean": 0.005514706019312143, "clip_ratio/low_min": 0.005514706019312143, "clip_ratio/high_mean": 0.011004327097907662, "clip_ratio/high_max": 0.011004327097907662, "clip_ratio/region_mean": 0.016519033117219806, "reward_total_mean": 0.9954667687416077, "reward_meter_mean": 0.9954667687416077, "reward_meter_std": 0.004938251804560423, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9954667687416077, "reward_total_composite_std": 0.004938251804560423} {"timestamp_utc": "2026-04-12T00:44:36Z", "mode": "train", "global_step": 1728, "epoch": 0.06940595252440053, "loss": 0.0319, "grad_norm": 5.639673709869385, "learning_rate": 4.766666666666667e-06, "num_tokens": 3892529.0, "completions/mean_length": 82.125, "completions/min_length": 72.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.125, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.9778019189834595, "rewards/meter/std": 0.024416539818048477, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9778019189834595, "rewards/total_composite/std": 0.024416539818048477, "reward": 0.9778019189834595, "reward_std": 0.024416528642177582, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.055459219962358475, "sampling/sampling_logp_difference/max": 1.5115740299224854, "sampling/importance_sampling_ratio/min": 0.31155264377593994, "sampling/importance_sampling_ratio/mean": 1.0134650468826294, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5294598899781704, "clip_ratio/low_mean": 0.005952381179668009, "clip_ratio/low_min": 0.005952381179668009, "clip_ratio/high_mean": 0.041200052946805954, "clip_ratio/high_max": 0.041200052946805954, "clip_ratio/region_mean": 0.04715243412647396, "reward_total_mean": 0.9778019189834595, "reward_meter_mean": 0.9778019189834595, "reward_meter_std": 0.024416539818048477, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9778019189834595, "reward_total_composite_std": 0.024416539818048477} {"timestamp_utc": "2026-04-12T00:44:43Z", "mode": "train", "global_step": 1729, "epoch": 0.06944611800618548, "loss": 0.0142, "grad_norm": 2.5409841537475586, "learning_rate": 4.763636363636364e-06, "num_tokens": 3895339.0, "completions/mean_length": 150.25, "completions/min_length": 135.0, "completions/max_length": 156.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 150.25, "completions/min_terminated_length": 135.0, "completions/max_terminated_length": 156.0, "rewards/meter/mean": 0.9943567514419556, "rewards/meter/std": 0.0023485159035772085, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8415178656578064, "rewards/repeat_penalty/std": 0.056821081787347794, "rewards/total_composite/mean": 0.836736798286438, "rewards/total_composite/std": 0.05605412647128105, "reward": 0.836736798286438, "reward_std": 0.05605413764715195, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016792375594377518, "sampling/sampling_logp_difference/max": 1.8594321012496948, "sampling/importance_sampling_ratio/min": 0.15576106309890747, "sampling/importance_sampling_ratio/mean": 1.0042328834533691, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09930440969765186, "clip_ratio/low_mean": 0.0008223684271797538, "clip_ratio/low_min": 0.0008223684271797538, "clip_ratio/high_mean": 0.01018475356977433, "clip_ratio/high_max": 0.01018475356977433, "clip_ratio/region_mean": 0.011007121996954083, "reward_total_mean": 0.836736798286438, "reward_meter_mean": 0.9943567514419556, "reward_meter_std": 0.0023485159035772085, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8415178656578064, "reward_repeat_penalty_std": 0.056821081787347794, "reward_total_composite_mean": 0.836736798286438, "reward_total_composite_std": 0.05605412647128105} {"timestamp_utc": "2026-04-12T00:44:50Z", "mode": "train", "global_step": 1730, "epoch": 0.06948628348797044, "loss": 0.0161, "grad_norm": 4.371915340423584, "learning_rate": 4.760606060606061e-06, "num_tokens": 3897799.0, "completions/mean_length": 131.5, "completions/min_length": 126.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.5, "completions/min_terminated_length": 126.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.9530513882637024, "rewards/meter/std": 0.020072180777788162, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6852272748947144, "rewards/repeat_penalty/std": 0.10432641208171844, "rewards/total_composite/mean": 0.6233780384063721, "rewards/total_composite/std": 0.12500165402889252, "reward": 0.6233780384063721, "reward_std": 0.12500165402889252, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030718868598341942, "sampling/sampling_logp_difference/max": 1.7257447242736816, "sampling/importance_sampling_ratio/min": 0.17804041504859924, "sampling/importance_sampling_ratio/mean": 1.0002104043960571, "sampling/importance_sampling_ratio/max": 1.8074078559875488, "entropy": 0.14938442967832088, "clip_ratio/low_mean": 0.013116939691826701, "clip_ratio/low_min": 0.013116939691826701, "clip_ratio/high_mean": 0.024344916339032352, "clip_ratio/high_max": 0.024344916339032352, "clip_ratio/region_mean": 0.03746185603085905, "reward_total_mean": 0.6233780384063721, "reward_meter_mean": 0.9530513882637024, "reward_meter_std": 0.020072180777788162, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6852272748947144, "reward_repeat_penalty_std": 0.10432641208171844, "reward_total_composite_mean": 0.6233780384063721, "reward_total_composite_std": 0.12500165402889252} {"timestamp_utc": "2026-04-12T00:44:56Z", "mode": "train", "global_step": 1731, "epoch": 0.06952644896975539, "loss": -0.0001, "grad_norm": 5.360625267028809, "learning_rate": 4.757575757575758e-06, "num_tokens": 3899601.0, "completions/mean_length": 68.25, "completions/min_length": 68.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.25, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.996942937374115, "rewards/meter/std": 0.0017319588223472238, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.996942937374115, "rewards/total_composite/std": 0.0017319588223472238, "reward": 0.996942937374115, "reward_std": 0.0017319645266979933, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017641214653849602, "sampling/sampling_logp_difference/max": 1.565748691558838, "sampling/importance_sampling_ratio/min": 0.20893153548240662, "sampling/importance_sampling_ratio/mean": 1.0010026693344116, "sampling/importance_sampling_ratio/max": 1.5452033281326294, "entropy": 0.08362119551748037, "clip_ratio/low_mean": 0.0036764706019312143, "clip_ratio/low_min": 0.0036764706019312143, "clip_ratio/high_mean": 0.007273018010891974, "clip_ratio/high_max": 0.007273018010891974, "clip_ratio/region_mean": 0.010949488612823188, "reward_total_mean": 0.996942937374115, "reward_meter_mean": 0.996942937374115, "reward_meter_std": 0.0017319588223472238, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.996942937374115, "reward_total_composite_std": 0.0017319588223472238} {"timestamp_utc": "2026-04-12T00:45:01Z", "mode": "train", "global_step": 1732, "epoch": 0.06956661445154035, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.754545454545455e-06, "num_tokens": 3901161.0, "completions/mean_length": 30.0, "completions/min_length": 30.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.0, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.992271900177002, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.992271900177002, "rewards/total_composite/std": 0.0, "reward": 0.992271900177002, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0011375559261068702, "sampling/sampling_logp_difference/max": 0.07951271533966064, "sampling/importance_sampling_ratio/min": 0.9235662817955017, "sampling/importance_sampling_ratio/mean": 1.0000195503234863, "sampling/importance_sampling_ratio/max": 1.0209813117980957, "entropy": 0.008589746314100921, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.992271900177002, "reward_meter_mean": 0.992271900177002, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.992271900177002, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T00:45:06Z", "mode": "train", "global_step": 1733, "epoch": 0.0696067799333253, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.751515151515152e-06, "num_tokens": 3902961.0, "completions/mean_length": 58.0, "completions/min_length": 58.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.993520200252533, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.993520200252533, "rewards/total_composite/std": 0.0, "reward": 0.993520200252533, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0015307251596823335, "sampling/sampling_logp_difference/max": 0.11124386638402939, "sampling/importance_sampling_ratio/min": 0.8947205543518066, "sampling/importance_sampling_ratio/mean": 1.000504970550537, "sampling/importance_sampling_ratio/max": 1.0491514205932617, "entropy": 0.012850302853621542, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.993520200252533, "reward_meter_mean": 0.993520200252533, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.993520200252533, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T00:45:10Z", "mode": "train", "global_step": 1734, "epoch": 0.06964694541511025, "loss": -0.0114, "grad_norm": 2.5078203678131104, "learning_rate": 4.748484848484849e-06, "num_tokens": 3904691.0, "completions/mean_length": 62.25, "completions/min_length": 60.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.25, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9855538606643677, "rewards/meter/std": 0.027445603162050247, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9855538606643677, "rewards/total_composite/std": 0.027445603162050247, "reward": 0.9855538606643677, "reward_std": 0.0274455975741148, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021761858835816383, "sampling/sampling_logp_difference/max": 1.1243009567260742, "sampling/importance_sampling_ratio/min": 0.3248794972896576, "sampling/importance_sampling_ratio/mean": 0.9950600266456604, "sampling/importance_sampling_ratio/max": 1.428898811340332, "entropy": 0.08833573758602142, "clip_ratio/low_mean": 0.0020833334419876337, "clip_ratio/low_min": 0.0020833334419876337, "clip_ratio/high_mean": 0.01626984216272831, "clip_ratio/high_max": 0.01626984216272831, "clip_ratio/region_mean": 0.018353175604715943, "reward_total_mean": 0.9855538606643677, "reward_meter_mean": 0.9855538606643677, "reward_meter_std": 0.027445603162050247, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9855538606643677, "reward_total_composite_std": 0.027445603162050247} {"timestamp_utc": "2026-04-12T00:45:14Z", "mode": "train", "global_step": 1735, "epoch": 0.06968711089689521, "loss": -0.0859, "grad_norm": 3.9663925170898438, "learning_rate": 4.745454545454546e-06, "num_tokens": 3906499.0, "completions/mean_length": 56.0, "completions/min_length": 42.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.0, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9921239614486694, "rewards/meter/std": 0.0039491597563028336, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9921239614486694, "rewards/total_composite/std": 0.0039491597563028336, "reward": 0.9921239614486694, "reward_std": 0.0039491597563028336, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006989335175603628, "sampling/sampling_logp_difference/max": 0.8487052917480469, "sampling/importance_sampling_ratio/min": 0.42796868085861206, "sampling/importance_sampling_ratio/mean": 1.000370740890503, "sampling/importance_sampling_ratio/max": 1.7702317237854004, "entropy": 0.037228189525194466, "clip_ratio/low_mean": 0.008928571827709675, "clip_ratio/low_min": 0.008928571827709675, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.008928571827709675, "reward_total_mean": 0.9921239614486694, "reward_meter_mean": 0.9921239614486694, "reward_meter_std": 0.0039491597563028336, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9921239614486694, "reward_total_composite_std": 0.0039491597563028336} {"timestamp_utc": "2026-04-12T00:45:19Z", "mode": "train", "global_step": 1736, "epoch": 0.06972727637868016, "loss": 0.03, "grad_norm": 5.453359127044678, "learning_rate": 4.7424242424242426e-06, "num_tokens": 3908212.0, "completions/mean_length": 63.125, "completions/min_length": 60.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.125, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.8737130761146545, "rewards/meter/std": 0.3438531160354614, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8737130761146545, "rewards/total_composite/std": 0.3438531160354614, "reward": 0.8737130761146545, "reward_std": 0.34385308623313904, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01911444216966629, "sampling/sampling_logp_difference/max": 0.978426456451416, "sampling/importance_sampling_ratio/min": 0.3759021461009979, "sampling/importance_sampling_ratio/mean": 1.003002643585205, "sampling/importance_sampling_ratio/max": 1.8557262420654297, "entropy": 0.12713485257700086, "clip_ratio/low_mean": 0.0037313431967049837, "clip_ratio/low_min": 0.0037313431967049837, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0037313431967049837, "reward_total_mean": 0.8737130761146545, "reward_meter_mean": 0.8737130761146545, "reward_meter_std": 0.3438531160354614, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8737130761146545, "reward_total_composite_std": 0.3438531160354614} {"timestamp_utc": "2026-04-12T00:45:24Z", "mode": "train", "global_step": 1737, "epoch": 0.06976744186046512, "loss": -0.0033, "grad_norm": 4.734118461608887, "learning_rate": 4.73939393939394e-06, "num_tokens": 3910546.0, "completions/mean_length": 129.75, "completions/min_length": 126.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 129.75, "completions/min_terminated_length": 126.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.8132379651069641, "rewards/meter/std": 0.2665981948375702, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8392857313156128, "rewards/repeat_penalty/std": 0.09155284613370895, "rewards/total_composite/mean": 0.6668680310249329, "rewards/total_composite/std": 0.21354439854621887, "reward": 0.6668680310249329, "reward_std": 0.21354439854621887, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02951814979314804, "sampling/sampling_logp_difference/max": 1.1711840629577637, "sampling/importance_sampling_ratio/min": 0.3099996745586395, "sampling/importance_sampling_ratio/mean": 1.0038542747497559, "sampling/importance_sampling_ratio/max": 1.6680244207382202, "entropy": 0.18008941877633333, "clip_ratio/low_mean": 0.008928571944124997, "clip_ratio/low_min": 0.008928571944124997, "clip_ratio/high_mean": 0.010484297177754343, "clip_ratio/high_max": 0.010484297177754343, "clip_ratio/region_mean": 0.01941286912187934, "reward_total_mean": 0.6668680310249329, "reward_meter_mean": 0.8132379651069641, "reward_meter_std": 0.2665981948375702, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8392857313156128, "reward_repeat_penalty_std": 0.09155284613370895, "reward_total_composite_mean": 0.6668680310249329, "reward_total_composite_std": 0.21354439854621887} {"timestamp_utc": "2026-04-12T00:45:30Z", "mode": "train", "global_step": 1738, "epoch": 0.06980760734225007, "loss": 0.0006, "grad_norm": 0.2504282295703888, "learning_rate": 4.736363636363637e-06, "num_tokens": 3913550.0, "completions/mean_length": 172.5, "completions/min_length": 169.0, "completions/max_length": 173.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 172.5, "completions/min_terminated_length": 169.0, "completions/max_terminated_length": 173.0, "rewards/meter/mean": 0.994848370552063, "rewards/meter/std": 0.00046264444245025516, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7124999761581421, "rewards/repeat_penalty/std": 0.0353553481400013, "rewards/total_composite/mean": 0.7088274955749512, "rewards/total_composite/std": 0.03512411192059517, "reward": 0.7088274955749512, "reward_std": 0.03512410447001457, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00417359871789813, "sampling/sampling_logp_difference/max": 1.570167064666748, "sampling/importance_sampling_ratio/min": 0.20801043510437012, "sampling/importance_sampling_ratio/mean": 0.9992320537567139, "sampling/importance_sampling_ratio/max": 1.0626802444458008, "entropy": 0.014313822146505117, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0014450866729021072, "clip_ratio/high_max": 0.0014450866729021072, "clip_ratio/region_mean": 0.0014450866729021072, "reward_total_mean": 0.7088274955749512, "reward_meter_mean": 0.994848370552063, "reward_meter_std": 0.00046264444245025516, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7124999761581421, "reward_repeat_penalty_std": 0.0353553481400013, "reward_total_composite_mean": 0.7088274955749512, "reward_total_composite_std": 0.03512411192059517} {"timestamp_utc": "2026-04-12T00:45:35Z", "mode": "train", "global_step": 1739, "epoch": 0.06984777282403502, "loss": -0.0102, "grad_norm": 4.318221092224121, "learning_rate": 4.7333333333333335e-06, "num_tokens": 3915644.0, "completions/mean_length": 81.75, "completions/min_length": 79.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.75, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.9966109991073608, "rewards/meter/std": 0.0017567714676260948, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9966109991073608, "rewards/total_composite/std": 0.0017567714676260948, "reward": 0.9966109991073608, "reward_std": 0.0017567694885656238, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04741675779223442, "sampling/sampling_logp_difference/max": 0.8447170257568359, "sampling/importance_sampling_ratio/min": 0.42967894673347473, "sampling/importance_sampling_ratio/mean": 1.0067728757858276, "sampling/importance_sampling_ratio/max": 1.6328699588775635, "entropy": 0.46590081229805946, "clip_ratio/low_mean": 0.006289557088166475, "clip_ratio/low_min": 0.006289557088166475, "clip_ratio/high_mean": 0.028822204330936074, "clip_ratio/high_max": 0.028822204330936074, "clip_ratio/region_mean": 0.03511176141910255, "reward_total_mean": 0.9966109991073608, "reward_meter_mean": 0.9966109991073608, "reward_meter_std": 0.0017567714676260948, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9966109991073608, "reward_total_composite_std": 0.0017567714676260948} {"timestamp_utc": "2026-04-12T00:45:39Z", "mode": "train", "global_step": 1740, "epoch": 0.06988793830581998, "loss": -0.009, "grad_norm": 7.030989646911621, "learning_rate": 4.730303030303031e-06, "num_tokens": 3917352.0, "completions/mean_length": 57.5, "completions/min_length": 55.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.5, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.8923461437225342, "rewards/meter/std": 0.2858184278011322, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8923461437225342, "rewards/total_composite/std": 0.2858184278011322, "reward": 0.8923461437225342, "reward_std": 0.2858184278011322, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006146615371108055, "sampling/sampling_logp_difference/max": 0.3172045946121216, "sampling/importance_sampling_ratio/min": 0.8774904012680054, "sampling/importance_sampling_ratio/mean": 1.0057528018951416, "sampling/importance_sampling_ratio/max": 1.3732835054397583, "entropy": 0.028293918527197093, "clip_ratio/low_mean": 0.00909090880304575, "clip_ratio/low_min": 0.00909090880304575, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.00909090880304575, "reward_total_mean": 0.8923461437225342, "reward_meter_mean": 0.8923461437225342, "reward_meter_std": 0.2858184278011322, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8923461437225342, "reward_total_composite_std": 0.2858184278011322} {"timestamp_utc": "2026-04-12T00:45:47Z", "mode": "train", "global_step": 1741, "epoch": 0.06992810378760493, "loss": 0.0527, "grad_norm": 1.3457516431808472, "learning_rate": 4.727272727272728e-06, "num_tokens": 3920951.0, "completions/mean_length": 273.875, "completions/min_length": 252.0, "completions/max_length": 290.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 273.875, "completions/min_terminated_length": 252.0, "completions/max_terminated_length": 290.0, "rewards/meter/mean": 0.9979905486106873, "rewards/meter/std": 0.0006236601620912552, "rewards/count_adherence/mean": 0.910714328289032, "rewards/count_adherence/std": 0.07393559068441391, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6429487466812134, "rewards/repeat_penalty/std": 0.04667491465806961, "rewards/total_composite/mean": 0.5869752168655396, "rewards/total_composite/std": 0.08794166892766953, "reward": 0.5869752168655396, "reward_std": 0.08794166892766953, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01812179759144783, "sampling/sampling_logp_difference/max": 5.0853352546691895, "sampling/importance_sampling_ratio/min": 0.006186812650412321, "sampling/importance_sampling_ratio/mean": 1.0031613111495972, "sampling/importance_sampling_ratio/max": 1.4954962730407715, "entropy": 0.11419002618640661, "clip_ratio/low_mean": 0.0034997850307263434, "clip_ratio/low_min": 0.0034997850307263434, "clip_ratio/high_mean": 0.004889143165200949, "clip_ratio/high_max": 0.004889143165200949, "clip_ratio/region_mean": 0.008388928195927292, "reward_total_mean": 0.5869752168655396, "reward_meter_mean": 0.9979905486106873, "reward_meter_std": 0.0006236601620912552, "reward_count_adherence_mean": 0.910714328289032, "reward_count_adherence_std": 0.07393559068441391, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6429487466812134, "reward_repeat_penalty_std": 0.04667491465806961, "reward_total_composite_mean": 0.5869752168655396, "reward_total_composite_std": 0.08794166892766953} {"timestamp_utc": "2026-04-12T00:45:52Z", "mode": "train", "global_step": 1742, "epoch": 0.06996826926938989, "loss": 0.0179, "grad_norm": 2.8153226375579834, "learning_rate": 4.724242424242424e-06, "num_tokens": 3923203.0, "completions/mean_length": 119.5, "completions/min_length": 114.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 119.5, "completions/min_terminated_length": 114.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.9969746470451355, "rewards/meter/std": 0.0006134378490969539, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.9471026659011841, "rewards/total_composite/std": 0.09206004440784454, "reward": 0.9471026659011841, "reward_std": 0.09206003695726395, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04116864502429962, "sampling/sampling_logp_difference/max": 1.3741235733032227, "sampling/importance_sampling_ratio/min": 0.25306129455566406, "sampling/importance_sampling_ratio/mean": 1.0121161937713623, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.397213451564312, "clip_ratio/low_mean": 0.0020493179326877, "clip_ratio/low_min": 0.0020493179326877, "clip_ratio/high_mean": 0.02945910906419158, "clip_ratio/high_max": 0.02945910906419158, "clip_ratio/region_mean": 0.03150842699687928, "reward_total_mean": 0.9471026659011841, "reward_meter_mean": 0.9969746470451355, "reward_meter_std": 0.0006134378490969539, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.9471026659011841, "reward_total_composite_std": 0.09206004440784454} {"timestamp_utc": "2026-04-12T00:45:57Z", "mode": "train", "global_step": 1743, "epoch": 0.07000843475117484, "loss": -0.0255, "grad_norm": 3.2874183654785156, "learning_rate": 4.721212121212122e-06, "num_tokens": 3925534.0, "completions/mean_length": 115.375, "completions/min_length": 114.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.375, "completions/min_terminated_length": 114.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.9939801692962646, "rewards/meter/std": 0.0004661441489588469, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.71875, "rewards/repeat_penalty/std": 0.012626901268959045, "rewards/total_composite/mean": 0.7144180536270142, "rewards/total_composite/std": 0.012203346937894821, "reward": 0.7144180536270142, "reward_std": 0.012203331105411053, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005304521415382624, "sampling/sampling_logp_difference/max": 2.7986814975738525, "sampling/importance_sampling_ratio/min": 0.06089029461145401, "sampling/importance_sampling_ratio/mean": 0.9994580745697021, "sampling/importance_sampling_ratio/max": 1.4329923391342163, "entropy": 0.009788335533812642, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0010000000474974513, "clip_ratio/high_max": 0.0010000000474974513, "clip_ratio/region_mean": 0.0010000000474974513, "reward_total_mean": 0.7144180536270142, "reward_meter_mean": 0.9939801692962646, "reward_meter_std": 0.0004661441489588469, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.71875, "reward_repeat_penalty_std": 0.012626901268959045, "reward_total_composite_mean": 0.7144180536270142, "reward_total_composite_std": 0.012203346937894821} {"timestamp_utc": "2026-04-12T00:46:02Z", "mode": "train", "global_step": 1744, "epoch": 0.07004860023295979, "loss": 0.0389, "grad_norm": 3.9866888523101807, "learning_rate": 4.718181818181818e-06, "num_tokens": 3927884.0, "completions/mean_length": 116.75, "completions/min_length": 111.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.75, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.9970704913139343, "rewards/meter/std": 0.0010195234790444374, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9970704913139343, "rewards/total_composite/std": 0.0010195234790444374, "reward": 0.9970704913139343, "reward_std": 0.0010194990318268538, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05034464970231056, "sampling/sampling_logp_difference/max": 1.4906072616577148, "sampling/importance_sampling_ratio/min": 0.22523583471775055, "sampling/importance_sampling_ratio/mean": 1.0086452960968018, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.44544921815395355, "clip_ratio/low_mean": 0.011764100869186223, "clip_ratio/low_min": 0.011764100869186223, "clip_ratio/high_mean": 0.024856957141309977, "clip_ratio/high_max": 0.024856957141309977, "clip_ratio/region_mean": 0.0366210580104962, "reward_total_mean": 0.9970704913139343, "reward_meter_mean": 0.9970704913139343, "reward_meter_std": 0.0010195234790444374, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9970704913139343, "reward_total_composite_std": 0.0010195234790444374} {"timestamp_utc": "2026-04-12T00:46:07Z", "mode": "train", "global_step": 1745, "epoch": 0.07008876571474475, "loss": 0.0183, "grad_norm": 5.250048637390137, "learning_rate": 4.715151515151515e-06, "num_tokens": 3929691.0, "completions/mean_length": 71.875, "completions/min_length": 65.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.875, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.8960655927658081, "rewards/meter/std": 0.09317556023597717, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8960655927658081, "rewards/total_composite/std": 0.09317556023597717, "reward": 0.8960655927658081, "reward_std": 0.09317556023597717, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04669762775301933, "sampling/sampling_logp_difference/max": 1.2144498825073242, "sampling/importance_sampling_ratio/min": 0.2968732714653015, "sampling/importance_sampling_ratio/mean": 1.0071378946304321, "sampling/importance_sampling_ratio/max": 1.7080049514770508, "entropy": 0.4467601887881756, "clip_ratio/low_mean": 0.01703576883301139, "clip_ratio/low_min": 0.01703576883301139, "clip_ratio/high_mean": 0.026507998118177056, "clip_ratio/high_max": 0.026507998118177056, "clip_ratio/region_mean": 0.043543766951188445, "reward_total_mean": 0.8960655927658081, "reward_meter_mean": 0.8960655927658081, "reward_meter_std": 0.09317556023597717, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8960655927658081, "reward_total_composite_std": 0.09317556023597717} {"timestamp_utc": "2026-04-12T00:46:12Z", "mode": "train", "global_step": 1746, "epoch": 0.0701289311965297, "loss": 0.0131, "grad_norm": 2.7793028354644775, "learning_rate": 4.7121212121212126e-06, "num_tokens": 3932034.0, "completions/mean_length": 118.875, "completions/min_length": 114.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.875, "completions/min_terminated_length": 114.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.9958000183105469, "rewards/meter/std": 0.0027918845880776644, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9958000183105469, "rewards/total_composite/std": 0.0027918845880776644, "reward": 0.9958000183105469, "reward_std": 0.0027918978594243526, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03678244352340698, "sampling/sampling_logp_difference/max": 1.1251811981201172, "sampling/importance_sampling_ratio/min": 0.3245936334133148, "sampling/importance_sampling_ratio/mean": 1.0113009214401245, "sampling/importance_sampling_ratio/max": 1.7213435173034668, "entropy": 0.3711636923253536, "clip_ratio/low_mean": 0.010288415476679802, "clip_ratio/low_min": 0.010288415476679802, "clip_ratio/high_mean": 0.01485425140708685, "clip_ratio/high_max": 0.01485425140708685, "clip_ratio/region_mean": 0.02514266688376665, "reward_total_mean": 0.9958000183105469, "reward_meter_mean": 0.9958000183105469, "reward_meter_std": 0.0027918845880776644, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9958000183105469, "reward_total_composite_std": 0.0027918845880776644} {"timestamp_utc": "2026-04-12T00:46:18Z", "mode": "train", "global_step": 1747, "epoch": 0.07016909667831465, "loss": 0.0052, "grad_norm": 2.528543710708618, "learning_rate": 4.709090909090909e-06, "num_tokens": 3934342.0, "completions/mean_length": 133.5, "completions/min_length": 130.0, "completions/max_length": 144.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 133.5, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 144.0, "rewards/meter/mean": 0.9185582995414734, "rewards/meter/std": 0.03270983695983887, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7857142686843872, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.7207491397857666, "rewards/total_composite/std": 0.0627276748418808, "reward": 0.7207491397857666, "reward_std": 0.0627276822924614, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018684184178709984, "sampling/sampling_logp_difference/max": 0.9996829032897949, "sampling/importance_sampling_ratio/min": 0.36799612641334534, "sampling/importance_sampling_ratio/mean": 1.0059216022491455, "sampling/importance_sampling_ratio/max": 1.7609641551971436, "entropy": 0.14490803238004446, "clip_ratio/low_mean": 0.005481442145537585, "clip_ratio/low_min": 0.005481442145537585, "clip_ratio/high_mean": 0.00946290313731879, "clip_ratio/high_max": 0.00946290313731879, "clip_ratio/region_mean": 0.014944345282856375, "reward_total_mean": 0.7207491397857666, "reward_meter_mean": 0.9185582995414734, "reward_meter_std": 0.03270983695983887, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7857142686843872, "reward_repeat_penalty_std": 0.07636035233736038, "reward_total_composite_mean": 0.7207491397857666, "reward_total_composite_std": 0.0627276748418808} {"timestamp_utc": "2026-04-12T00:46:23Z", "mode": "train", "global_step": 1748, "epoch": 0.07020926216009961, "loss": 0.0045, "grad_norm": 5.013108253479004, "learning_rate": 4.706060606060606e-06, "num_tokens": 3936075.0, "completions/mean_length": 70.625, "completions/min_length": 67.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.625, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9558204412460327, "rewards/meter/std": 0.03730607032775879, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9558204412460327, "rewards/total_composite/std": 0.03730607032775879, "reward": 0.9558204412460327, "reward_std": 0.03730607405304909, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.055451519787311554, "sampling/sampling_logp_difference/max": 1.048119068145752, "sampling/importance_sampling_ratio/min": 0.3505965769290924, "sampling/importance_sampling_ratio/mean": 1.006996989250183, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45604707673192024, "clip_ratio/low_mean": 0.01255695940926671, "clip_ratio/low_min": 0.01255695940926671, "clip_ratio/high_mean": 0.02681165118701756, "clip_ratio/high_max": 0.02681165118701756, "clip_ratio/region_mean": 0.03936861059628427, "reward_total_mean": 0.9558204412460327, "reward_meter_mean": 0.9558204412460327, "reward_meter_std": 0.03730607032775879, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9558204412460327, "reward_total_composite_std": 0.03730607032775879} {"timestamp_utc": "2026-04-12T00:46:28Z", "mode": "train", "global_step": 1749, "epoch": 0.07024942764188456, "loss": 0.0118, "grad_norm": 6.071586608886719, "learning_rate": 4.7030303030303035e-06, "num_tokens": 3937998.0, "completions/mean_length": 76.375, "completions/min_length": 73.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.375, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9766583442687988, "rewards/meter/std": 0.02674211747944355, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9766583442687988, "rewards/total_composite/std": 0.02674211747944355, "reward": 0.9766583442687988, "reward_std": 0.02674211747944355, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08423186093568802, "sampling/sampling_logp_difference/max": 2.003994941711426, "sampling/importance_sampling_ratio/min": 0.13479571044445038, "sampling/importance_sampling_ratio/mean": 1.0096642971038818, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6718878448009491, "clip_ratio/low_mean": 0.023322351276874542, "clip_ratio/low_min": 0.023322351276874542, "clip_ratio/high_mean": 0.048262338852509856, "clip_ratio/high_max": 0.048262338852509856, "clip_ratio/region_mean": 0.0715846901293844, "reward_total_mean": 0.9766583442687988, "reward_meter_mean": 0.9766583442687988, "reward_meter_std": 0.02674211747944355, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9766583442687988, "reward_total_composite_std": 0.02674211747944355} {"timestamp_utc": "2026-04-12T00:46:33Z", "mode": "train", "global_step": 1750, "epoch": 0.07028959312366952, "loss": -0.0126, "grad_norm": 2.597658157348633, "learning_rate": 4.7e-06, "num_tokens": 3939868.0, "completions/mean_length": 74.75, "completions/min_length": 72.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.75, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.998544454574585, "rewards/meter/std": 0.00028903057682327926, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998544454574585, "rewards/total_composite/std": 0.00028903057682327926, "reward": 0.998544454574585, "reward_std": 0.00028902888880111277, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04236089065670967, "sampling/sampling_logp_difference/max": 1.5543270111083984, "sampling/importance_sampling_ratio/min": 0.2113315463066101, "sampling/importance_sampling_ratio/mean": 1.0116918087005615, "sampling/importance_sampling_ratio/max": 1.9856641292572021, "entropy": 0.3266411516815424, "clip_ratio/low_mean": 0.018771880073472857, "clip_ratio/low_min": 0.018771880073472857, "clip_ratio/high_mean": 0.021297869854606688, "clip_ratio/high_max": 0.021297869854606688, "clip_ratio/region_mean": 0.040069749928079545, "reward_total_mean": 0.998544454574585, "reward_meter_mean": 0.998544454574585, "reward_meter_std": 0.00028903057682327926, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.998544454574585, "reward_total_composite_std": 0.00028903057682327926} {"timestamp_utc": "2026-04-12T00:47:39Z", "mode": "eval", "global_step": 1750, "epoch": 0.07028959312366952, "eval_loss": NaN, "eval_runtime": 66.4368, "eval_samples_per_second": 1.565, "eval_steps_per_second": 0.196, "eval_num_tokens": 3939868.0, "eval_completions/mean_length": 198.26923076923077, "eval_completions/min_length": 63.15384615384615, "eval_completions/max_length": 347.38461538461536, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 198.26923076923077, "eval_completions/min_terminated_length": 63.15384615384615, "eval_completions/max_terminated_length": 347.38461538461536, "eval_rewards/meter/mean": 0.7541542970217191, "eval_rewards/meter/std": 0.3719308634216969, "eval_rewards/count_adherence/mean": 0.9028738003510696, "eval_rewards/count_adherence/std": 0.12786896412189191, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/repeat_penalty/mean": 0.8075039799396808, "eval_rewards/repeat_penalty/std": 0.16454027822384468, "eval_rewards/total_composite/mean": 0.5515764791231889, "eval_rewards/total_composite/std": 0.33034826585879695, "eval_reward": 0.5515764791231889, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.018031520339158866, "eval_sampling/sampling_logp_difference/max": 1.1275318952707143, "eval_sampling/importance_sampling_ratio/min": 0.33156666732751405, "eval_sampling/importance_sampling_ratio/mean": 1.004565248122582, "eval_sampling/importance_sampling_ratio/max": 1.4309009680381188, "eval_entropy": 0.1877427167044236, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5515764791231889, "eval_reward_meter_mean": 0.7541542970217191, "eval_reward_meter_std": 0.3719308634216969, "eval_reward_count_adherence_mean": 0.9028738003510696, "eval_reward_count_adherence_std": 0.12786896412189191, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_repeat_penalty_mean": 0.8075039799396808, "eval_reward_repeat_penalty_std": 0.16454027822384468, "eval_reward_total_composite_mean": 0.5515764791231889, "eval_reward_total_composite_std": 0.33034826585879695} {"timestamp_utc": "2026-04-12T00:47:47Z", "mode": "train", "global_step": 1751, "epoch": 0.07032975860545447, "loss": -0.0145, "grad_norm": 3.8397207260131836, "learning_rate": 4.696969696969698e-06, "num_tokens": 3941579.0, "completions/mean_length": 59.875, "completions/min_length": 58.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.875, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.994328498840332, "rewards/meter/std": 0.0016656159423291683, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.994328498840332, "rewards/total_composite/std": 0.0016656159423291683, "reward": 0.994328498840332, "reward_std": 0.001665610121563077, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02080354280769825, "sampling/sampling_logp_difference/max": 0.939934492111206, "sampling/importance_sampling_ratio/min": 0.39065343141555786, "sampling/importance_sampling_ratio/mean": 1.0038416385650635, "sampling/importance_sampling_ratio/max": 1.5397275686264038, "entropy": 0.12970400508493185, "clip_ratio/low_mean": 0.00628531095571816, "clip_ratio/low_min": 0.00628531095571816, "clip_ratio/high_mean": 0.006016385043039918, "clip_ratio/high_max": 0.006016385043039918, "clip_ratio/region_mean": 0.012301695998758078, "reward_total_mean": 0.994328498840332, "reward_meter_mean": 0.994328498840332, "reward_meter_std": 0.0016656159423291683, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.994328498840332, "reward_total_composite_std": 0.0016656159423291683} {"timestamp_utc": "2026-04-12T00:47:52Z", "mode": "train", "global_step": 1752, "epoch": 0.07036992408723942, "loss": 0.0264, "grad_norm": 9.45784854888916, "learning_rate": 4.693939393939394e-06, "num_tokens": 3943517.0, "completions/mean_length": 78.25, "completions/min_length": 70.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.25, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.9296289086341858, "rewards/meter/std": 0.09625125676393509, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9296289086341858, "rewards/total_composite/std": 0.09625125676393509, "reward": 0.9296289086341858, "reward_std": 0.09625126421451569, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08540250360965729, "sampling/sampling_logp_difference/max": 1.4676272869110107, "sampling/importance_sampling_ratio/min": 0.230471670627594, "sampling/importance_sampling_ratio/mean": 1.0267534255981445, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7139590308070183, "clip_ratio/low_mean": 0.02355617005378008, "clip_ratio/low_min": 0.02355617005378008, "clip_ratio/high_mean": 0.04827680857852101, "clip_ratio/high_max": 0.04827680857852101, "clip_ratio/region_mean": 0.07183297863230109, "reward_total_mean": 0.9296289086341858, "reward_meter_mean": 0.9296289086341858, "reward_meter_std": 0.09625125676393509, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9296289086341858, "reward_total_composite_std": 0.09625125676393509} {"timestamp_utc": "2026-04-12T00:47:56Z", "mode": "train", "global_step": 1753, "epoch": 0.07041008956902438, "loss": -0.0159, "grad_norm": 3.8942646980285645, "learning_rate": 4.690909090909092e-06, "num_tokens": 3945216.0, "completions/mean_length": 61.375, "completions/min_length": 59.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.375, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9958833456039429, "rewards/meter/std": 0.0009836297249421477, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9958833456039429, "rewards/total_composite/std": 0.0009836297249421477, "reward": 0.9958833456039429, "reward_std": 0.0009836361277848482, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01771390438079834, "sampling/sampling_logp_difference/max": 1.0666460990905762, "sampling/importance_sampling_ratio/min": 0.34416088461875916, "sampling/importance_sampling_ratio/mean": 1.0048010349273682, "sampling/importance_sampling_ratio/max": 1.43393075466156, "entropy": 0.11703337356448174, "clip_ratio/low_mean": 0.008269546087831259, "clip_ratio/low_min": 0.008269546087831259, "clip_ratio/high_mean": 0.005952381296083331, "clip_ratio/high_max": 0.005952381296083331, "clip_ratio/region_mean": 0.01422192738391459, "reward_total_mean": 0.9958833456039429, "reward_meter_mean": 0.9958833456039429, "reward_meter_std": 0.0009836297249421477, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9958833456039429, "reward_total_composite_std": 0.0009836297249421477} {"timestamp_utc": "2026-04-12T00:48:01Z", "mode": "train", "global_step": 1754, "epoch": 0.07045025505080933, "loss": 0.0065, "grad_norm": 3.0640108585357666, "learning_rate": 4.687878787878788e-06, "num_tokens": 3947246.0, "completions/mean_length": 73.75, "completions/min_length": 72.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.75, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.998239278793335, "rewards/meter/std": 0.0011535151861608028, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998239278793335, "rewards/total_composite/std": 0.0011535151861608028, "reward": 0.998239278793335, "reward_std": 0.0011535151861608028, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.050084829330444336, "sampling/sampling_logp_difference/max": 1.3049192428588867, "sampling/importance_sampling_ratio/min": 0.2711944580078125, "sampling/importance_sampling_ratio/mean": 1.0054799318313599, "sampling/importance_sampling_ratio/max": 1.8171170949935913, "entropy": 0.4345889072865248, "clip_ratio/low_mean": 0.015182648552581668, "clip_ratio/low_min": 0.015182648552581668, "clip_ratio/high_mean": 0.030779469525441527, "clip_ratio/high_max": 0.030779469525441527, "clip_ratio/region_mean": 0.045962118078023195, "reward_total_mean": 0.998239278793335, "reward_meter_mean": 0.998239278793335, "reward_meter_std": 0.0011535151861608028, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.998239278793335, "reward_total_composite_std": 0.0011535151861608028} {"timestamp_utc": "2026-04-12T00:48:06Z", "mode": "train", "global_step": 1755, "epoch": 0.07049042053259429, "loss": 0.0099, "grad_norm": 1.097916841506958, "learning_rate": 4.684848484848485e-06, "num_tokens": 3949239.0, "completions/mean_length": 68.125, "completions/min_length": 67.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.125, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9972657561302185, "rewards/meter/std": 0.0007608237792737782, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9972657561302185, "rewards/total_composite/std": 0.0007608237792737782, "reward": 0.9972657561302185, "reward_std": 0.0007608263404108584, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009788503870368004, "sampling/sampling_logp_difference/max": 0.9967951774597168, "sampling/importance_sampling_ratio/min": 0.7174111604690552, "sampling/importance_sampling_ratio/mean": 1.0076888799667358, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06952884793281555, "clip_ratio/low_mean": 0.0053571430034935474, "clip_ratio/low_min": 0.0053571430034935474, "clip_ratio/high_mean": 0.0036764706019312143, "clip_ratio/high_max": 0.0036764706019312143, "clip_ratio/region_mean": 0.009033613605424762, "reward_total_mean": 0.9972657561302185, "reward_meter_mean": 0.9972657561302185, "reward_meter_std": 0.0007608237792737782, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9972657561302185, "reward_total_composite_std": 0.0007608237792737782} {"timestamp_utc": "2026-04-12T00:48:11Z", "mode": "train", "global_step": 1756, "epoch": 0.07053058601437924, "loss": 0.013, "grad_norm": 5.3303751945495605, "learning_rate": 4.681818181818183e-06, "num_tokens": 3950770.0, "completions/mean_length": 35.375, "completions/min_length": 35.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.375, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9960504770278931, "rewards/meter/std": 0.0025489823892712593, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9960504770278931, "rewards/total_composite/std": 0.0025489823892712593, "reward": 0.9960504770278931, "reward_std": 0.0025490045081824064, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026823023334145546, "sampling/sampling_logp_difference/max": 1.4001922607421875, "sampling/importance_sampling_ratio/min": 0.2465495616197586, "sampling/importance_sampling_ratio/mean": 1.0033628940582275, "sampling/importance_sampling_ratio/max": 1.6020153760910034, "entropy": 0.14098990987986326, "clip_ratio/low_mean": 0.01736111124046147, "clip_ratio/low_min": 0.01736111124046147, "clip_ratio/high_mean": 0.010714285774156451, "clip_ratio/high_max": 0.010714285774156451, "clip_ratio/region_mean": 0.02807539701461792, "reward_total_mean": 0.9960504770278931, "reward_meter_mean": 0.9960504770278931, "reward_meter_std": 0.0025489823892712593, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9960504770278931, "reward_total_composite_std": 0.0025489823892712593} {"timestamp_utc": "2026-04-12T00:48:16Z", "mode": "train", "global_step": 1757, "epoch": 0.0705707514961642, "loss": 0.0099, "grad_norm": 2.692962884902954, "learning_rate": 4.678787878787879e-06, "num_tokens": 3953321.0, "completions/mean_length": 148.875, "completions/min_length": 145.0, "completions/max_length": 153.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 148.875, "completions/min_terminated_length": 145.0, "completions/max_terminated_length": 153.0, "rewards/meter/mean": 0.9979263544082642, "rewards/meter/std": 0.0016821667086333036, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.09155284613370895, "rewards/total_composite/mean": 0.8732737898826599, "rewards/total_composite/std": 0.09220181405544281, "reward": 0.8732737898826599, "reward_std": 0.09220179915428162, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04017876833677292, "sampling/sampling_logp_difference/max": 1.1506986618041992, "sampling/importance_sampling_ratio/min": 0.3164156377315521, "sampling/importance_sampling_ratio/mean": 1.0055673122406006, "sampling/importance_sampling_ratio/max": 1.883940577507019, "entropy": 0.3133071381598711, "clip_ratio/low_mean": 0.020188873866572976, "clip_ratio/low_min": 0.020188873866572976, "clip_ratio/high_mean": 0.010034454986453056, "clip_ratio/high_max": 0.010034454986453056, "clip_ratio/region_mean": 0.030223328853026032, "reward_total_mean": 0.8732737898826599, "reward_meter_mean": 0.9979263544082642, "reward_meter_std": 0.0016821667086333036, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.09155284613370895, "reward_total_composite_mean": 0.8732737898826599, "reward_total_composite_std": 0.09220181405544281} {"timestamp_utc": "2026-04-12T00:48:21Z", "mode": "train", "global_step": 1758, "epoch": 0.07061091697794915, "loss": 0.0195, "grad_norm": 6.611409664154053, "learning_rate": 4.675757575757576e-06, "num_tokens": 3955117.0, "completions/mean_length": 76.5, "completions/min_length": 75.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.5, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9615824222564697, "rewards/meter/std": 0.03308320790529251, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9615824222564697, "rewards/total_composite/std": 0.03308320790529251, "reward": 0.9615824222564697, "reward_std": 0.03308321535587311, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06566065549850464, "sampling/sampling_logp_difference/max": 1.2004508972167969, "sampling/importance_sampling_ratio/min": 0.30105844140052795, "sampling/importance_sampling_ratio/mean": 1.0291359424591064, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7143527865409851, "clip_ratio/low_mean": 0.02743737120181322, "clip_ratio/low_min": 0.02743737120181322, "clip_ratio/high_mean": 0.03448549238964915, "clip_ratio/high_max": 0.03448549238964915, "clip_ratio/region_mean": 0.061922863591462374, "reward_total_mean": 0.9615824222564697, "reward_meter_mean": 0.9615824222564697, "reward_meter_std": 0.03308320790529251, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9615824222564697, "reward_total_composite_std": 0.03308320790529251} {"timestamp_utc": "2026-04-12T00:48:25Z", "mode": "train", "global_step": 1759, "epoch": 0.0706510824597341, "loss": 0.0438, "grad_norm": 5.4466166496276855, "learning_rate": 4.6727272727272735e-06, "num_tokens": 3956703.0, "completions/mean_length": 36.25, "completions/min_length": 35.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.25, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.9933856725692749, "rewards/meter/std": 0.007373345550149679, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9933856725692749, "rewards/total_composite/std": 0.007373345550149679, "reward": 0.9933856725692749, "reward_std": 0.007373336236923933, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021740470081567764, "sampling/sampling_logp_difference/max": 0.7743911743164062, "sampling/importance_sampling_ratio/min": 0.46098437905311584, "sampling/importance_sampling_ratio/mean": 1.0038979053497314, "sampling/importance_sampling_ratio/max": 1.7787117958068848, "entropy": 0.12411041604354978, "clip_ratio/low_mean": 0.016025641234591603, "clip_ratio/low_min": 0.016025641234591603, "clip_ratio/high_mean": 0.0034722222480922937, "clip_ratio/high_max": 0.0034722222480922937, "clip_ratio/region_mean": 0.019497863482683897, "reward_total_mean": 0.9933856725692749, "reward_meter_mean": 0.9933856725692749, "reward_meter_std": 0.007373345550149679, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9933856725692749, "reward_total_composite_std": 0.007373345550149679} {"timestamp_utc": "2026-04-12T00:48:31Z", "mode": "train", "global_step": 1760, "epoch": 0.07069124794151906, "loss": -0.0871, "grad_norm": 4.831878662109375, "learning_rate": 4.66969696969697e-06, "num_tokens": 3959470.0, "completions/mean_length": 141.875, "completions/min_length": 125.0, "completions/max_length": 152.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 141.875, "completions/min_terminated_length": 125.0, "completions/max_terminated_length": 152.0, "rewards/meter/mean": 0.992239236831665, "rewards/meter/std": 0.0005570180364884436, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.1035098284482956, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.609375, "rewards/repeat_penalty/std": 0.012938717380166054, "rewards/total_composite/mean": 0.558111310005188, "rewards/total_composite/std": 0.05107751861214638, "reward": 0.558111310005188, "reward_std": 0.051077522337436676, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008833160623908043, "sampling/sampling_logp_difference/max": 2.1678309440612793, "sampling/importance_sampling_ratio/min": 0.11442553997039795, "sampling/importance_sampling_ratio/mean": 0.9995898008346558, "sampling/importance_sampling_ratio/max": 1.4811135530471802, "entropy": 0.02644984540529549, "clip_ratio/low_mean": 0.0020000000949949026, "clip_ratio/low_min": 0.0020000000949949026, "clip_ratio/high_mean": 0.003289473708719015, "clip_ratio/high_max": 0.003289473708719015, "clip_ratio/region_mean": 0.005289473803713918, "reward_total_mean": 0.558111310005188, "reward_meter_mean": 0.992239236831665, "reward_meter_std": 0.0005570180364884436, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.1035098284482956, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.609375, "reward_repeat_penalty_std": 0.012938717380166054, "reward_total_composite_mean": 0.558111310005188, "reward_total_composite_std": 0.05107751861214638} {"timestamp_utc": "2026-04-12T00:48:35Z", "mode": "train", "global_step": 1761, "epoch": 0.07073141342330401, "loss": -0.0004, "grad_norm": 0.03243406489491463, "learning_rate": 4.666666666666667e-06, "num_tokens": 3961206.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9981468319892883, "rewards/meter/std": 7.291422207345022e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981468319892883, "rewards/total_composite/std": 7.291422207345022e-06, "reward": 0.9981468319892883, "reward_std": 7.28539862393518e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0019470350816845894, "sampling/sampling_logp_difference/max": 0.4015388488769531, "sampling/importance_sampling_ratio/min": 0.8816590905189514, "sampling/importance_sampling_ratio/mean": 1.0011060237884521, "sampling/importance_sampling_ratio/max": 1.4941221475601196, "entropy": 0.01341856678482145, "clip_ratio/low_mean": 0.001953125, "clip_ratio/low_min": 0.001953125, "clip_ratio/high_mean": 0.001953125, "clip_ratio/high_max": 0.001953125, "clip_ratio/region_mean": 0.00390625, "reward_total_mean": 0.9981468319892883, "reward_meter_mean": 0.9981468319892883, "reward_meter_std": 7.291422207345022e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981468319892883, "reward_total_composite_std": 7.291422207345022e-06} {"timestamp_utc": "2026-04-12T00:48:40Z", "mode": "train", "global_step": 1762, "epoch": 0.07077157890508896, "loss": 0.0019, "grad_norm": 0.9674805998802185, "learning_rate": 4.663636363636364e-06, "num_tokens": 3963328.0, "completions/mean_length": 95.25, "completions/min_length": 95.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.25, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.9970584511756897, "rewards/meter/std": 0.00035686453338712454, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.8475098013877869, "rewards/total_composite/std": 0.09242907166481018, "reward": 0.8475098013877869, "reward_std": 0.09242908656597137, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00739458529278636, "sampling/sampling_logp_difference/max": 0.7102556228637695, "sampling/importance_sampling_ratio/min": 0.4915185272693634, "sampling/importance_sampling_ratio/mean": 1.0031148195266724, "sampling/importance_sampling_ratio/max": 1.3529959917068481, "entropy": 0.04998829588294029, "clip_ratio/low_mean": 0.0026178729021921754, "clip_ratio/low_min": 0.0026178729021921754, "clip_ratio/high_mean": 0.0013157895300537348, "clip_ratio/high_max": 0.0013157895300537348, "clip_ratio/region_mean": 0.00393366243224591, "reward_total_mean": 0.8475098013877869, "reward_meter_mean": 0.9970584511756897, "reward_meter_std": 0.00035686453338712454, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.8475098013877869, "reward_total_composite_std": 0.09242907166481018} {"timestamp_utc": "2026-04-12T00:48:48Z", "mode": "train", "global_step": 1763, "epoch": 0.07081174438687392, "loss": -0.0018, "grad_norm": 1.530092477798462, "learning_rate": 4.660606060606061e-06, "num_tokens": 3967632.0, "completions/mean_length": 331.0, "completions/min_length": 316.0, "completions/max_length": 347.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 331.0, "completions/min_terminated_length": 316.0, "completions/max_terminated_length": 347.0, "rewards/meter/mean": 0.9972841739654541, "rewards/meter/std": 0.0010668373433873057, "rewards/count_adherence/mean": 0.692307710647583, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8308823108673096, "rewards/repeat_penalty/std": 0.06623479723930359, "rewards/total_composite/mean": 0.5736789703369141, "rewards/total_composite/std": 0.04592465981841087, "reward": 0.5736789703369141, "reward_std": 0.04592467471957207, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030713124200701714, "sampling/sampling_logp_difference/max": 1.0928246974945068, "sampling/importance_sampling_ratio/min": 0.33526813983917236, "sampling/importance_sampling_ratio/mean": 1.0089269876480103, "sampling/importance_sampling_ratio/max": 1.8286789655685425, "entropy": 0.308476896956563, "clip_ratio/low_mean": 0.013588423724286258, "clip_ratio/low_min": 0.013588423724286258, "clip_ratio/high_mean": 0.011261471780017018, "clip_ratio/high_max": 0.011261471780017018, "clip_ratio/region_mean": 0.024849895504303277, "reward_total_mean": 0.5736789703369141, "reward_meter_mean": 0.9972841739654541, "reward_meter_std": 0.0010668373433873057, "reward_count_adherence_mean": 0.692307710647583, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8308823108673096, "reward_repeat_penalty_std": 0.06623479723930359, "reward_total_composite_mean": 0.5736789703369141, "reward_total_composite_std": 0.04592465981841087} {"timestamp_utc": "2026-04-12T00:48:52Z", "mode": "train", "global_step": 1764, "epoch": 0.07085190986865887, "loss": -0.0047, "grad_norm": 5.83869743347168, "learning_rate": 4.657575757575758e-06, "num_tokens": 3969088.0, "completions/mean_length": 35.0, "completions/min_length": 34.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9969139099121094, "rewards/meter/std": 0.002184208482503891, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9969139099121094, "rewards/total_composite/std": 0.002184208482503891, "reward": 0.9969139099121094, "reward_std": 0.002184197073802352, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01693633943796158, "sampling/sampling_logp_difference/max": 0.8935699462890625, "sampling/importance_sampling_ratio/min": 0.40919235348701477, "sampling/importance_sampling_ratio/mean": 0.9999939203262329, "sampling/importance_sampling_ratio/max": 1.648281455039978, "entropy": 0.09531538654118776, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9969139099121094, "reward_meter_mean": 0.9969139099121094, "reward_meter_std": 0.002184208482503891, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9969139099121094, "reward_total_composite_std": 0.002184208482503891} {"timestamp_utc": "2026-04-12T00:49:01Z", "mode": "train", "global_step": 1765, "epoch": 0.07089207535044383, "loss": -0.0026, "grad_norm": 1.4083707332611084, "learning_rate": 4.654545454545455e-06, "num_tokens": 3973789.0, "completions/mean_length": 400.625, "completions/min_length": 381.0, "completions/max_length": 426.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 400.625, "completions/min_terminated_length": 381.0, "completions/max_terminated_length": 426.0, "rewards/meter/mean": 0.9959809184074402, "rewards/meter/std": 0.002331522526219487, "rewards/count_adherence/mean": 0.6176470518112183, "rewards/count_adherence/std": 0.03144249692559242, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8089442253112793, "rewards/repeat_penalty/std": 0.07961878180503845, "rewards/total_composite/mean": 0.4966247081756592, "rewards/total_composite/std": 0.04498037323355675, "reward": 0.4966247081756592, "reward_std": 0.04498036950826645, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030164016410708427, "sampling/sampling_logp_difference/max": 1.2716550827026367, "sampling/importance_sampling_ratio/min": 0.2803671956062317, "sampling/importance_sampling_ratio/mean": 1.0090073347091675, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.30641947500407696, "clip_ratio/low_mean": 0.013535281177610159, "clip_ratio/low_min": 0.013535281177610159, "clip_ratio/high_mean": 0.00853747595101595, "clip_ratio/high_max": 0.00853747595101595, "clip_ratio/region_mean": 0.022072757128626108, "reward_total_mean": 0.4966247081756592, "reward_meter_mean": 0.9959809184074402, "reward_meter_std": 0.002331522526219487, "reward_count_adherence_mean": 0.6176470518112183, "reward_count_adherence_std": 0.03144249692559242, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8089442253112793, "reward_repeat_penalty_std": 0.07961878180503845, "reward_total_composite_mean": 0.4966247081756592, "reward_total_composite_std": 0.04498037323355675} {"timestamp_utc": "2026-04-12T00:49:06Z", "mode": "train", "global_step": 1766, "epoch": 0.07093224083222878, "loss": 0.0029, "grad_norm": 3.6053273677825928, "learning_rate": 4.651515151515152e-06, "num_tokens": 3976025.0, "completions/mean_length": 110.5, "completions/min_length": 107.0, "completions/max_length": 113.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 110.5, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 113.0, "rewards/meter/mean": 0.9980085492134094, "rewards/meter/std": 0.0012038704007863998, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.9480756521224976, "rewards/total_composite/std": 0.09206356108188629, "reward": 0.9480756521224976, "reward_std": 0.09206356853246689, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04403796046972275, "sampling/sampling_logp_difference/max": 1.918081283569336, "sampling/importance_sampling_ratio/min": 0.14688852429389954, "sampling/importance_sampling_ratio/mean": 1.0053863525390625, "sampling/importance_sampling_ratio/max": 1.5807472467422485, "entropy": 0.38782429322600365, "clip_ratio/low_mean": 0.00802111136727035, "clip_ratio/low_min": 0.00802111136727035, "clip_ratio/high_mean": 0.03843654436059296, "clip_ratio/high_max": 0.03843654436059296, "clip_ratio/region_mean": 0.04645765572786331, "reward_total_mean": 0.9480756521224976, "reward_meter_mean": 0.9980085492134094, "reward_meter_std": 0.0012038704007863998, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.9480756521224976, "reward_total_composite_std": 0.09206356108188629} {"timestamp_utc": "2026-04-12T00:49:10Z", "mode": "train", "global_step": 1767, "epoch": 0.07097240631401373, "loss": 0.0007, "grad_norm": 0.5792075991630554, "learning_rate": 4.648484848484849e-06, "num_tokens": 3977817.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9981694221496582, "rewards/meter/std": 9.33830306166783e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981694221496582, "rewards/total_composite/std": 9.33830306166783e-05, "reward": 0.9981694221496582, "reward_std": 9.336639777757227e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006093635223805904, "sampling/sampling_logp_difference/max": 0.6290621757507324, "sampling/importance_sampling_ratio/min": 0.5330914855003357, "sampling/importance_sampling_ratio/mean": 0.9992143511772156, "sampling/importance_sampling_ratio/max": 1.7626099586486816, "entropy": 0.018734007026068866, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.00390625, "reward_total_mean": 0.9981694221496582, "reward_meter_mean": 0.9981694221496582, "reward_meter_std": 9.33830306166783e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981694221496582, "reward_total_composite_std": 9.33830306166783e-05} {"timestamp_utc": "2026-04-12T00:49:15Z", "mode": "train", "global_step": 1768, "epoch": 0.07101257179579869, "loss": 0.0077, "grad_norm": 4.614588737487793, "learning_rate": 4.645454545454545e-06, "num_tokens": 3979649.0, "completions/mean_length": 68.0, "completions/min_length": 68.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.0, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9972659349441528, "rewards/meter/std": 0.0006626376416534185, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9972659349441528, "rewards/total_composite/std": 0.0006626376416534185, "reward": 0.9972659349441528, "reward_std": 0.0006626351969316602, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020455962046980858, "sampling/sampling_logp_difference/max": 0.9924750328063965, "sampling/importance_sampling_ratio/min": 0.37065815925598145, "sampling/importance_sampling_ratio/mean": 1.0017422437667847, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11503921588882804, "clip_ratio/low_mean": 0.005514706019312143, "clip_ratio/low_min": 0.005514706019312143, "clip_ratio/high_mean": 0.009191176504828036, "clip_ratio/high_max": 0.009191176504828036, "clip_ratio/region_mean": 0.01470588252414018, "reward_total_mean": 0.9972659349441528, "reward_meter_mean": 0.9972659349441528, "reward_meter_std": 0.0006626376416534185, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9972659349441528, "reward_total_composite_std": 0.0006626376416534185} {"timestamp_utc": "2026-04-12T00:49:19Z", "mode": "train", "global_step": 1769, "epoch": 0.07105273727758364, "loss": 0.0015, "grad_norm": 1.5824861526489258, "learning_rate": 4.642424242424243e-06, "num_tokens": 3981601.0, "completions/mean_length": 63.0, "completions/min_length": 63.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.0, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9967877864837646, "rewards/meter/std": 0.00022891137632541358, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9967877864837646, "rewards/total_composite/std": 0.00022891137632541358, "reward": 0.9967877864837646, "reward_std": 0.00022891460685059428, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00849367305636406, "sampling/sampling_logp_difference/max": 0.3904110789299011, "sampling/importance_sampling_ratio/min": 0.6767785549163818, "sampling/importance_sampling_ratio/mean": 1.003195881843567, "sampling/importance_sampling_ratio/max": 1.2106847763061523, "entropy": 0.060156718362122774, "clip_ratio/low_mean": 0.003968254197388887, "clip_ratio/low_min": 0.003968254197388887, "clip_ratio/high_mean": 0.005952381296083331, "clip_ratio/high_max": 0.005952381296083331, "clip_ratio/region_mean": 0.009920635493472219, "reward_total_mean": 0.9967877864837646, "reward_meter_mean": 0.9967877864837646, "reward_meter_std": 0.00022891137632541358, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9967877864837646, "reward_total_composite_std": 0.00022891137632541358} {"timestamp_utc": "2026-04-12T00:49:24Z", "mode": "train", "global_step": 1770, "epoch": 0.0710929027593686, "loss": 0.1371, "grad_norm": 7.558566093444824, "learning_rate": 4.63939393939394e-06, "num_tokens": 3983404.0, "completions/mean_length": 81.375, "completions/min_length": 73.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.375, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.9489094018936157, "rewards/meter/std": 0.09742892533540726, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.886616051197052, "rewards/total_composite/std": 0.18366359174251556, "reward": 0.886616051197052, "reward_std": 0.18366359174251556, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06127572059631348, "sampling/sampling_logp_difference/max": 1.5139522552490234, "sampling/importance_sampling_ratio/min": 0.22003860771656036, "sampling/importance_sampling_ratio/mean": 1.0119951963424683, "sampling/importance_sampling_ratio/max": 1.8940720558166504, "entropy": 0.5354950428009033, "clip_ratio/low_mean": 0.012420599116012454, "clip_ratio/low_min": 0.012420599116012454, "clip_ratio/high_mean": 0.029366800910793245, "clip_ratio/high_max": 0.029366800910793245, "clip_ratio/region_mean": 0.0417874000268057, "reward_total_mean": 0.886616051197052, "reward_meter_mean": 0.9489094018936157, "reward_meter_std": 0.09742892533540726, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.886616051197052, "reward_total_composite_std": 0.18366359174251556} {"timestamp_utc": "2026-04-12T00:49:29Z", "mode": "train", "global_step": 1771, "epoch": 0.07113306824115355, "loss": 0.006, "grad_norm": 2.8212172985076904, "learning_rate": 4.636363636363636e-06, "num_tokens": 3985659.0, "completions/mean_length": 94.875, "completions/min_length": 94.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.875, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.9980397820472717, "rewards/meter/std": 0.0001069462377927266, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.8732877969741821, "rewards/total_composite/std": 0.10333773493766785, "reward": 0.8732877969741821, "reward_std": 0.10333773493766785, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009597206488251686, "sampling/sampling_logp_difference/max": 0.9711115956306458, "sampling/importance_sampling_ratio/min": 0.3786618709564209, "sampling/importance_sampling_ratio/mean": 0.9989795684814453, "sampling/importance_sampling_ratio/max": 1.4593793153762817, "entropy": 0.03504605917260051, "clip_ratio/low_mean": 0.003947368590161204, "clip_ratio/low_min": 0.003947368590161204, "clip_ratio/high_mean": 0.0066069429740309715, "clip_ratio/high_max": 0.0066069429740309715, "clip_ratio/region_mean": 0.010554311564192176, "reward_total_mean": 0.8732877969741821, "reward_meter_mean": 0.9980397820472717, "reward_meter_std": 0.0001069462377927266, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.8732877969741821, "reward_total_composite_std": 0.10333773493766785} {"timestamp_utc": "2026-04-12T00:49:33Z", "mode": "train", "global_step": 1772, "epoch": 0.0711732337229385, "loss": -0.0013, "grad_norm": 9.031041145324707, "learning_rate": 4.633333333333334e-06, "num_tokens": 3987488.0, "completions/mean_length": 62.625, "completions/min_length": 60.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.625, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9955183267593384, "rewards/meter/std": 0.0038252200465649366, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9955183267593384, "rewards/total_composite/std": 0.0038252200465649366, "reward": 0.9955183267593384, "reward_std": 0.0038252256345003843, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02198421023786068, "sampling/sampling_logp_difference/max": 1.470834732055664, "sampling/importance_sampling_ratio/min": 0.2297336310148239, "sampling/importance_sampling_ratio/mean": 0.9984506964683533, "sampling/importance_sampling_ratio/max": 1.5951910018920898, "entropy": 0.10218251217156649, "clip_ratio/low_mean": 0.0020833334419876337, "clip_ratio/low_min": 0.0020833334419876337, "clip_ratio/high_mean": 0.007936508394777775, "clip_ratio/high_max": 0.007936508394777775, "clip_ratio/region_mean": 0.010019841836765409, "reward_total_mean": 0.9955183267593384, "reward_meter_mean": 0.9955183267593384, "reward_meter_std": 0.0038252200465649366, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9955183267593384, "reward_total_composite_std": 0.0038252200465649366} {"timestamp_utc": "2026-04-12T00:49:40Z", "mode": "train", "global_step": 1773, "epoch": 0.07121339920472346, "loss": 0.0489, "grad_norm": 2.2303996086120605, "learning_rate": 4.630303030303031e-06, "num_tokens": 3990562.0, "completions/mean_length": 200.25, "completions/min_length": 191.0, "completions/max_length": 223.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 200.25, "completions/min_terminated_length": 191.0, "completions/max_terminated_length": 223.0, "rewards/meter/mean": 0.9357019662857056, "rewards/meter/std": 0.17194922268390656, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.07715168595314026, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7526223659515381, "rewards/repeat_penalty/std": 0.07428380101919174, "rewards/total_composite/mean": 0.6658661961555481, "rewards/total_composite/std": 0.11478758603334427, "reward": 0.6658661961555481, "reward_std": 0.11478757858276367, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01970170997083187, "sampling/sampling_logp_difference/max": 2.9212241172790527, "sampling/importance_sampling_ratio/min": 0.053867705166339874, "sampling/importance_sampling_ratio/mean": 1.0023951530456543, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11565544595941901, "clip_ratio/low_mean": 0.006157553056254983, "clip_ratio/low_min": 0.006157553056254983, "clip_ratio/high_mean": 0.005890052416361868, "clip_ratio/high_max": 0.005890052416361868, "clip_ratio/region_mean": 0.012047605472616851, "reward_total_mean": 0.6658661961555481, "reward_meter_mean": 0.9357019662857056, "reward_meter_std": 0.17194922268390656, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.07715168595314026, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7526223659515381, "reward_repeat_penalty_std": 0.07428380101919174, "reward_total_composite_mean": 0.6658661961555481, "reward_total_composite_std": 0.11478758603334427} {"timestamp_utc": "2026-04-12T00:49:45Z", "mode": "train", "global_step": 1774, "epoch": 0.07125356468650841, "loss": 0.0009, "grad_norm": 2.5007715225219727, "learning_rate": 4.627272727272727e-06, "num_tokens": 3992891.0, "completions/mean_length": 127.125, "completions/min_length": 127.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.125, "completions/min_terminated_length": 127.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.997309684753418, "rewards/meter/std": 0.00012155534204794094, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8392857313156128, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.8370258808135986, "rewards/total_composite/std": 0.05034034699201584, "reward": 0.8370258808135986, "reward_std": 0.05034033954143524, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00933446828275919, "sampling/sampling_logp_difference/max": 1.754063606262207, "sampling/importance_sampling_ratio/min": 0.17306923866271973, "sampling/importance_sampling_ratio/mean": 1.0009863376617432, "sampling/importance_sampling_ratio/max": 1.2224539518356323, "entropy": 0.05814863136038184, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0019685039296746254, "clip_ratio/high_max": 0.0019685039296746254, "clip_ratio/region_mean": 0.0019685039296746254, "reward_total_mean": 0.8370258808135986, "reward_meter_mean": 0.997309684753418, "reward_meter_std": 0.00012155534204794094, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8392857313156128, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.8370258808135986, "reward_total_composite_std": 0.05034034699201584} {"timestamp_utc": "2026-04-12T00:49:50Z", "mode": "train", "global_step": 1775, "epoch": 0.07129373016829336, "loss": -0.0231, "grad_norm": 6.361879348754883, "learning_rate": 4.6242424242424245e-06, "num_tokens": 3994961.0, "completions/mean_length": 75.75, "completions/min_length": 71.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.75, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9477696418762207, "rewards/meter/std": 0.10307468473911285, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9477696418762207, "rewards/total_composite/std": 0.10307468473911285, "reward": 0.9477696418762207, "reward_std": 0.10307466983795166, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07908248156309128, "sampling/sampling_logp_difference/max": 3.3076226711273193, "sampling/importance_sampling_ratio/min": 0.036603085696697235, "sampling/importance_sampling_ratio/mean": 1.0081671476364136, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6292642988264561, "clip_ratio/low_mean": 0.007042253389954567, "clip_ratio/low_min": 0.007042253389954567, "clip_ratio/high_mean": 0.040617201710119843, "clip_ratio/high_max": 0.040617201710119843, "clip_ratio/region_mean": 0.04765945510007441, "reward_total_mean": 0.9477696418762207, "reward_meter_mean": 0.9477696418762207, "reward_meter_std": 0.10307468473911285, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9477696418762207, "reward_total_composite_std": 0.10307468473911285} {"timestamp_utc": "2026-04-12T00:49:54Z", "mode": "train", "global_step": 1776, "epoch": 0.07133389565007832, "loss": -0.0062, "grad_norm": 9.041850090026855, "learning_rate": 4.621212121212122e-06, "num_tokens": 3996754.0, "completions/mean_length": 56.125, "completions/min_length": 56.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.125, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.8854995965957642, "rewards/meter/std": 0.043195273727178574, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8854995965957642, "rewards/total_composite/std": 0.043195273727178574, "reward": 0.8854995965957642, "reward_std": 0.04319525882601738, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007866953499615192, "sampling/sampling_logp_difference/max": 1.9500985145568848, "sampling/importance_sampling_ratio/min": 0.7719552516937256, "sampling/importance_sampling_ratio/mean": 1.0033729076385498, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.028014210518449545, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8854995965957642, "reward_meter_mean": 0.8854995965957642, "reward_meter_std": 0.043195273727178574, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8854995965957642, "reward_total_composite_std": 0.043195273727178574} {"timestamp_utc": "2026-04-12T00:49:59Z", "mode": "train", "global_step": 1777, "epoch": 0.07137406113186327, "loss": 0.013, "grad_norm": 4.8217244148254395, "learning_rate": 4.618181818181818e-06, "num_tokens": 3998604.0, "completions/mean_length": 71.25, "completions/min_length": 69.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.25, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9572109580039978, "rewards/meter/std": 0.052123501896858215, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9572109580039978, "rewards/total_composite/std": 0.052123501896858215, "reward": 0.9572109580039978, "reward_std": 0.05212350934743881, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.057501811534166336, "sampling/sampling_logp_difference/max": 1.8455209732055664, "sampling/importance_sampling_ratio/min": 0.1579430252313614, "sampling/importance_sampling_ratio/mean": 1.0080149173736572, "sampling/importance_sampling_ratio/max": 1.790594458580017, "entropy": 0.4883102737367153, "clip_ratio/low_mean": 0.012251984560862184, "clip_ratio/low_min": 0.012251984560862184, "clip_ratio/high_mean": 0.03510406916029751, "clip_ratio/high_max": 0.03510406916029751, "clip_ratio/region_mean": 0.0473560537211597, "reward_total_mean": 0.9572109580039978, "reward_meter_mean": 0.9572109580039978, "reward_meter_std": 0.052123501896858215, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9572109580039978, "reward_total_composite_std": 0.052123501896858215} {"timestamp_utc": "2026-04-12T00:50:05Z", "mode": "train", "global_step": 1778, "epoch": 0.07141422661364823, "loss": 0.0316, "grad_norm": 4.067593097686768, "learning_rate": 4.615151515151515e-06, "num_tokens": 4001210.0, "completions/mean_length": 160.75, "completions/min_length": 146.0, "completions/max_length": 206.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 160.75, "completions/min_terminated_length": 146.0, "completions/max_terminated_length": 206.0, "rewards/meter/mean": 0.7919793128967285, "rewards/meter/std": 0.23337964713573456, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9665178656578064, "rewards/repeat_penalty/std": 0.062180306762456894, "rewards/total_composite/mean": 0.7290425300598145, "rewards/total_composite/std": 0.19308261573314667, "reward": 0.7290425300598145, "reward_std": 0.19308261573314667, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08566864579916, "sampling/sampling_logp_difference/max": 8.59061336517334, "sampling/importance_sampling_ratio/min": 0.00018584205827210099, "sampling/importance_sampling_ratio/mean": 1.014849066734314, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7425523065030575, "clip_ratio/low_mean": 0.017992550507187843, "clip_ratio/low_min": 0.017992550507187843, "clip_ratio/high_mean": 0.04102262854576111, "clip_ratio/high_max": 0.04102262854576111, "clip_ratio/region_mean": 0.05901517905294895, "reward_total_mean": 0.7290425300598145, "reward_meter_mean": 0.7919793128967285, "reward_meter_std": 0.23337964713573456, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9665178656578064, "reward_repeat_penalty_std": 0.062180306762456894, "reward_total_composite_mean": 0.7290425300598145, "reward_total_composite_std": 0.19308261573314667} {"timestamp_utc": "2026-04-12T00:50:10Z", "mode": "train", "global_step": 1779, "epoch": 0.07145439209543318, "loss": 0.0089, "grad_norm": 2.5793981552124023, "learning_rate": 4.612121212121212e-06, "num_tokens": 4003558.0, "completions/mean_length": 134.5, "completions/min_length": 133.0, "completions/max_length": 138.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 134.5, "completions/min_terminated_length": 133.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.9970629215240479, "rewards/meter/std": 0.0009571582195349038, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.854625403881073, "rewards/total_composite/std": 0.0008204205660149455, "reward": 0.854625403881073, "reward_std": 0.0008204328478313982, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010115544311702251, "sampling/sampling_logp_difference/max": 0.8139721155166626, "sampling/importance_sampling_ratio/min": 0.44309455156326294, "sampling/importance_sampling_ratio/mean": 1.0006462335586548, "sampling/importance_sampling_ratio/max": 1.5591769218444824, "entropy": 0.052033907268196344, "clip_ratio/low_mean": 0.0027173913549631834, "clip_ratio/low_min": 0.0027173913549631834, "clip_ratio/high_mean": 0.0074559845379553735, "clip_ratio/high_max": 0.0074559845379553735, "clip_ratio/region_mean": 0.010173375892918557, "reward_total_mean": 0.854625403881073, "reward_meter_mean": 0.9970629215240479, "reward_meter_std": 0.0009571582195349038, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.854625403881073, "reward_total_composite_std": 0.0008204205660149455} {"timestamp_utc": "2026-04-12T00:50:18Z", "mode": "train", "global_step": 1780, "epoch": 0.07149455757721813, "loss": 0.0069, "grad_norm": 2.3572487831115723, "learning_rate": 4.60909090909091e-06, "num_tokens": 4007791.0, "completions/mean_length": 304.125, "completions/min_length": 290.0, "completions/max_length": 313.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 304.125, "completions/min_terminated_length": 290.0, "completions/max_terminated_length": 313.0, "rewards/meter/mean": 0.9952999353408813, "rewards/meter/std": 0.0025832315441221, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.03928370773792267, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9177083373069763, "rewards/repeat_penalty/std": 0.08480106294155121, "rewards/total_composite/mean": 0.7979733347892761, "rewards/total_composite/std": 0.06831313669681549, "reward": 0.7979733347892761, "reward_std": 0.06831313669681549, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.045330848544836044, "sampling/sampling_logp_difference/max": 2.610581398010254, "sampling/importance_sampling_ratio/min": 0.07349180430173874, "sampling/importance_sampling_ratio/mean": 1.010035753250122, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4535849690437317, "clip_ratio/low_mean": 0.01071572583168745, "clip_ratio/low_min": 0.01071572583168745, "clip_ratio/high_mean": 0.018438909435644746, "clip_ratio/high_max": 0.018438909435644746, "clip_ratio/region_mean": 0.029154635267332196, "reward_total_mean": 0.7979733347892761, "reward_meter_mean": 0.9952999353408813, "reward_meter_std": 0.0025832315441221, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.03928370773792267, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9177083373069763, "reward_repeat_penalty_std": 0.08480106294155121, "reward_total_composite_mean": 0.7979733347892761, "reward_total_composite_std": 0.06831313669681549} {"timestamp_utc": "2026-04-12T00:50:22Z", "mode": "train", "global_step": 1781, "epoch": 0.07153472305900309, "loss": -0.0006, "grad_norm": 2.1961991786956787, "learning_rate": 4.606060606060606e-06, "num_tokens": 4009447.0, "completions/mean_length": 50.0, "completions/min_length": 50.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.0, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.9375852346420288, "rewards/meter/std": 0.0008122828439809382, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9375852346420288, "rewards/total_composite/std": 0.0008122828439809382, "reward": 0.9375852346420288, "reward_std": 0.0008122779545374215, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011180485598742962, "sampling/sampling_logp_difference/max": 0.815919816493988, "sampling/importance_sampling_ratio/min": 0.4422323405742645, "sampling/importance_sampling_ratio/mean": 1.0008137226104736, "sampling/importance_sampling_ratio/max": 1.5095360279083252, "entropy": 0.08982300292700529, "clip_ratio/low_mean": 0.0024999999441206455, "clip_ratio/low_min": 0.0024999999441206455, "clip_ratio/high_mean": 0.0024999999441206455, "clip_ratio/high_max": 0.0024999999441206455, "clip_ratio/region_mean": 0.004999999888241291, "reward_total_mean": 0.9375852346420288, "reward_meter_mean": 0.9375852346420288, "reward_meter_std": 0.0008122828439809382, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9375852346420288, "reward_total_composite_std": 0.0008122828439809382} {"timestamp_utc": "2026-04-12T00:50:26Z", "mode": "train", "global_step": 1782, "epoch": 0.07157488854078804, "loss": -0.0032, "grad_norm": 2.5923538208007812, "learning_rate": 4.603030303030304e-06, "num_tokens": 4011350.0, "completions/mean_length": 80.875, "completions/min_length": 79.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.875, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.9967532753944397, "rewards/meter/std": 0.0014706900110468268, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8722769021987915, "rewards/total_composite/std": 0.3524559736251831, "reward": 0.8722769021987915, "reward_std": 0.3524559438228607, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05614684522151947, "sampling/sampling_logp_difference/max": 1.5454130172729492, "sampling/importance_sampling_ratio/min": 0.21322380006313324, "sampling/importance_sampling_ratio/mean": 1.0133930444717407, "sampling/importance_sampling_ratio/max": 1.5346218347549438, "entropy": 0.5954937487840652, "clip_ratio/low_mean": 0.004746835213154554, "clip_ratio/low_min": 0.004746835213154554, "clip_ratio/high_mean": 0.03219789918512106, "clip_ratio/high_max": 0.03219789918512106, "clip_ratio/region_mean": 0.036944734398275614, "reward_total_mean": 0.8722769021987915, "reward_meter_mean": 0.9967532753944397, "reward_meter_std": 0.0014706900110468268, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8722769021987915, "reward_total_composite_std": 0.3524559736251831} {"timestamp_utc": "2026-04-12T00:50:32Z", "mode": "train", "global_step": 1783, "epoch": 0.071615054022573, "loss": 0.0012, "grad_norm": 0.8668001294136047, "learning_rate": 4.600000000000001e-06, "num_tokens": 4014214.0, "completions/mean_length": 152.0, "completions/min_length": 152.0, "completions/max_length": 152.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 152.0, "completions/min_terminated_length": 152.0, "completions/max_terminated_length": 152.0, "rewards/meter/mean": 0.9920245409011841, "rewards/meter/std": 0.00013670595944859087, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.612500011920929, "rewards/repeat_penalty/std": 0.035355325788259506, "rewards/total_composite/mean": 0.6076171398162842, "rewards/total_composite/std": 0.035118672996759415, "reward": 0.6076171398162842, "reward_std": 0.03511865437030792, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007640105206519365, "sampling/sampling_logp_difference/max": 2.3813014030456543, "sampling/importance_sampling_ratio/min": 0.09243020415306091, "sampling/importance_sampling_ratio/mean": 0.9984169006347656, "sampling/importance_sampling_ratio/max": 1.160125970840454, "entropy": 0.024257719283923507, "clip_ratio/low_mean": 0.004111842135898769, "clip_ratio/low_min": 0.004111842135898769, "clip_ratio/high_mean": 0.0008223684271797538, "clip_ratio/high_max": 0.0008223684271797538, "clip_ratio/region_mean": 0.004934210563078523, "reward_total_mean": 0.6076171398162842, "reward_meter_mean": 0.9920245409011841, "reward_meter_std": 0.00013670595944859087, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.612500011920929, "reward_repeat_penalty_std": 0.035355325788259506, "reward_total_composite_mean": 0.6076171398162842, "reward_total_composite_std": 0.035118672996759415} {"timestamp_utc": "2026-04-12T00:50:36Z", "mode": "train", "global_step": 1784, "epoch": 0.07165521950435795, "loss": 0.0087, "grad_norm": 4.6875691413879395, "learning_rate": 4.596969696969697e-06, "num_tokens": 4015926.0, "completions/mean_length": 65.0, "completions/min_length": 62.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.8023771643638611, "rewards/meter/std": 0.2034316211938858, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8023771643638611, "rewards/total_composite/std": 0.2034316211938858, "reward": 0.8023771643638611, "reward_std": 0.203431636095047, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0649370476603508, "sampling/sampling_logp_difference/max": 1.2312498092651367, "sampling/importance_sampling_ratio/min": 0.2919274866580963, "sampling/importance_sampling_ratio/mean": 1.0129239559173584, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5869800895452499, "clip_ratio/low_mean": 0.01740056835114956, "clip_ratio/low_min": 0.01740056835114956, "clip_ratio/high_mean": 0.02884695027023554, "clip_ratio/high_max": 0.02884695027023554, "clip_ratio/region_mean": 0.0462475186213851, "reward_total_mean": 0.8023771643638611, "reward_meter_mean": 0.8023771643638611, "reward_meter_std": 0.2034316211938858, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8023771643638611, "reward_total_composite_std": 0.2034316211938858} {"timestamp_utc": "2026-04-12T00:50:41Z", "mode": "train", "global_step": 1785, "epoch": 0.0716953849861429, "loss": -0.0151, "grad_norm": 2.411994457244873, "learning_rate": 4.5939393939393945e-06, "num_tokens": 4017713.0, "completions/mean_length": 62.375, "completions/min_length": 59.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.375, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9964926242828369, "rewards/meter/std": 0.0008941100095398724, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9964926242828369, "rewards/total_composite/std": 0.0008941100095398724, "reward": 0.9964926242828369, "reward_std": 0.0008941145497374237, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01294564176350832, "sampling/sampling_logp_difference/max": 1.444681167602539, "sampling/importance_sampling_ratio/min": 0.23582126200199127, "sampling/importance_sampling_ratio/mean": 1.0034265518188477, "sampling/importance_sampling_ratio/max": 1.3098853826522827, "entropy": 0.08539297711104155, "clip_ratio/low_mean": 0.004167824285104871, "clip_ratio/low_min": 0.004167824285104871, "clip_ratio/high_mean": 0.005921379197388887, "clip_ratio/high_max": 0.005921379197388887, "clip_ratio/region_mean": 0.010089203482493758, "reward_total_mean": 0.9964926242828369, "reward_meter_mean": 0.9964926242828369, "reward_meter_std": 0.0008941100095398724, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9964926242828369, "reward_total_composite_std": 0.0008941100095398724} {"timestamp_utc": "2026-04-12T00:50:47Z", "mode": "train", "global_step": 1786, "epoch": 0.07173555046792786, "loss": -0.0127, "grad_norm": 2.4926347732543945, "learning_rate": 4.590909090909092e-06, "num_tokens": 4020442.0, "completions/mean_length": 152.125, "completions/min_length": 151.0, "completions/max_length": 160.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 152.125, "completions/min_terminated_length": 151.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.9910025596618652, "rewards/meter/std": 0.0011056133080273867, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6749999523162842, "rewards/repeat_penalty/std": 0.103509820997715, "rewards/total_composite/mean": 0.6688380241394043, "rewards/total_composite/std": 0.10174151510000229, "reward": 0.6688380241394043, "reward_std": 0.1017415001988411, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008964202366769314, "sampling/sampling_logp_difference/max": 1.3915038108825684, "sampling/importance_sampling_ratio/min": 0.24870102107524872, "sampling/importance_sampling_ratio/mean": 1.0039024353027344, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.04743872885592282, "clip_ratio/low_mean": 0.0016556291375309229, "clip_ratio/low_min": 0.0016556291375309229, "clip_ratio/high_mean": 0.004780629184097052, "clip_ratio/high_max": 0.004780629184097052, "clip_ratio/region_mean": 0.0064362583216279745, "reward_total_mean": 0.6688380241394043, "reward_meter_mean": 0.9910025596618652, "reward_meter_std": 0.0011056133080273867, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6749999523162842, "reward_repeat_penalty_std": 0.103509820997715, "reward_total_composite_mean": 0.6688380241394043, "reward_total_composite_std": 0.10174151510000229} {"timestamp_utc": "2026-04-12T00:50:52Z", "mode": "train", "global_step": 1787, "epoch": 0.07177571594971281, "loss": 0.0007, "grad_norm": 1.5127758979797363, "learning_rate": 4.587878787878788e-06, "num_tokens": 4022570.0, "completions/mean_length": 95.0, "completions/min_length": 94.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.0, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.9981739521026611, "rewards/meter/std": 6.629144627368078e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9732200503349304, "rewards/total_composite/std": 0.07058726996183395, "reward": 0.9732200503349304, "reward_std": 0.07058726251125336, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010238762944936752, "sampling/sampling_logp_difference/max": 1.506108283996582, "sampling/importance_sampling_ratio/min": 0.22177138924598694, "sampling/importance_sampling_ratio/mean": 1.000563383102417, "sampling/importance_sampling_ratio/max": 1.5315839052200317, "entropy": 0.03921856731176376, "clip_ratio/low_mean": 0.003947368357330561, "clip_ratio/low_min": 0.003947368357330561, "clip_ratio/high_mean": 0.005263741128146648, "clip_ratio/high_max": 0.005263741128146648, "clip_ratio/region_mean": 0.009211109485477209, "reward_total_mean": 0.9732200503349304, "reward_meter_mean": 0.9981739521026611, "reward_meter_std": 6.629144627368078e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9732200503349304, "reward_total_composite_std": 0.07058726996183395} {"timestamp_utc": "2026-04-12T00:50:58Z", "mode": "train", "global_step": 1788, "epoch": 0.07181588143149777, "loss": 0.0073, "grad_norm": 2.8437070846557617, "learning_rate": 4.5848484848484854e-06, "num_tokens": 4025609.0, "completions/mean_length": 193.875, "completions/min_length": 190.0, "completions/max_length": 198.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 193.875, "completions/min_terminated_length": 190.0, "completions/max_terminated_length": 198.0, "rewards/meter/mean": 0.9956002235412598, "rewards/meter/std": 0.0012398899998515844, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.9679561853408813, "rewards/total_composite/std": 0.05145645886659622, "reward": 0.9679561853408813, "reward_std": 0.05145645514130592, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05295930430293083, "sampling/sampling_logp_difference/max": 1.3258538246154785, "sampling/importance_sampling_ratio/min": 0.2655761241912842, "sampling/importance_sampling_ratio/mean": 1.008656620979309, "sampling/importance_sampling_ratio/max": 1.9465110301971436, "entropy": 0.5489388853311539, "clip_ratio/low_mean": 0.0070457912515848875, "clip_ratio/low_min": 0.0070457912515848875, "clip_ratio/high_mean": 0.038041369058191776, "clip_ratio/high_max": 0.038041369058191776, "clip_ratio/region_mean": 0.045087160309776664, "reward_total_mean": 0.9679561853408813, "reward_meter_mean": 0.9956002235412598, "reward_meter_std": 0.0012398899998515844, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_total_composite_mean": 0.9679561853408813, "reward_total_composite_std": 0.05145645886659622} {"timestamp_utc": "2026-04-12T00:51:02Z", "mode": "train", "global_step": 1789, "epoch": 0.07185604691328272, "loss": 0.0114, "grad_norm": 3.5395383834838867, "learning_rate": 4.581818181818183e-06, "num_tokens": 4027414.0, "completions/mean_length": 67.625, "completions/min_length": 64.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.625, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9397947192192078, "rewards/meter/std": 0.026229292154312134, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9397947192192078, "rewards/total_composite/std": 0.026229292154312134, "reward": 0.9397947192192078, "reward_std": 0.026229292154312134, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04796122387051582, "sampling/sampling_logp_difference/max": 1.1403248310089111, "sampling/importance_sampling_ratio/min": 0.3197151720523834, "sampling/importance_sampling_ratio/mean": 1.0007556676864624, "sampling/importance_sampling_ratio/max": 1.5871034860610962, "entropy": 0.38035739958286285, "clip_ratio/low_mean": 0.011368778301402926, "clip_ratio/low_min": 0.011368778301402926, "clip_ratio/high_mean": 0.016516561503522098, "clip_ratio/high_max": 0.016516561503522098, "clip_ratio/region_mean": 0.027885339804925025, "reward_total_mean": 0.9397947192192078, "reward_meter_mean": 0.9397947192192078, "reward_meter_std": 0.026229292154312134, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9397947192192078, "reward_total_composite_std": 0.026229292154312134} {"timestamp_utc": "2026-04-12T00:51:08Z", "mode": "train", "global_step": 1790, "epoch": 0.07189621239506767, "loss": 0.0094, "grad_norm": 2.5047359466552734, "learning_rate": 4.578787878787879e-06, "num_tokens": 4029603.0, "completions/mean_length": 110.625, "completions/min_length": 107.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 110.625, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.9982960224151611, "rewards/meter/std": 0.000550406810361892, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9982960224151611, "rewards/total_composite/std": 0.000550406810361892, "reward": 0.9982960224151611, "reward_std": 0.0005504076834768057, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.048308465629816055, "sampling/sampling_logp_difference/max": 1.0485877990722656, "sampling/importance_sampling_ratio/min": 0.35043227672576904, "sampling/importance_sampling_ratio/mean": 1.005360722541809, "sampling/importance_sampling_ratio/max": 1.5884928703308105, "entropy": 0.4972764067351818, "clip_ratio/low_mean": 0.019280804554000497, "clip_ratio/low_min": 0.019280804554000497, "clip_ratio/high_mean": 0.013679954456165433, "clip_ratio/high_max": 0.013679954456165433, "clip_ratio/region_mean": 0.03296075901016593, "reward_total_mean": 0.9982960224151611, "reward_meter_mean": 0.9982960224151611, "reward_meter_std": 0.000550406810361892, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9982960224151611, "reward_total_composite_std": 0.000550406810361892} {"timestamp_utc": "2026-04-12T00:51:15Z", "mode": "train", "global_step": 1791, "epoch": 0.07193637787685263, "loss": -0.0086, "grad_norm": 2.072072982788086, "learning_rate": 4.575757575757576e-06, "num_tokens": 4033409.0, "completions/mean_length": 280.75, "completions/min_length": 270.0, "completions/max_length": 292.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 280.75, "completions/min_terminated_length": 270.0, "completions/max_terminated_length": 292.0, "rewards/meter/mean": 0.9915696978569031, "rewards/meter/std": 0.01817433536052704, "rewards/count_adherence/mean": 0.921875, "rewards/count_adherence/std": 0.06469365209341049, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8535714149475098, "rewards/repeat_penalty/std": 0.07463330775499344, "rewards/total_composite/mean": 0.7775777578353882, "rewards/total_composite/std": 0.05556074157357216, "reward": 0.7775777578353882, "reward_std": 0.05556074529886246, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03709190711379051, "sampling/sampling_logp_difference/max": 1.9197864532470703, "sampling/importance_sampling_ratio/min": 0.14663827419281006, "sampling/importance_sampling_ratio/mean": 1.0069139003753662, "sampling/importance_sampling_ratio/max": 1.9100106954574585, "entropy": 0.39633841440081596, "clip_ratio/low_mean": 0.009515275072772056, "clip_ratio/low_min": 0.009515275072772056, "clip_ratio/high_mean": 0.015547977527603507, "clip_ratio/high_max": 0.015547977527603507, "clip_ratio/region_mean": 0.025063252600375563, "reward_total_mean": 0.7775777578353882, "reward_meter_mean": 0.9915696978569031, "reward_meter_std": 0.01817433536052704, "reward_count_adherence_mean": 0.921875, "reward_count_adherence_std": 0.06469365209341049, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8535714149475098, "reward_repeat_penalty_std": 0.07463330775499344, "reward_total_composite_mean": 0.7775777578353882, "reward_total_composite_std": 0.05556074157357216} {"timestamp_utc": "2026-04-12T00:51:20Z", "mode": "train", "global_step": 1792, "epoch": 0.07197654335863758, "loss": 0.0012, "grad_norm": 0.36933040618896484, "learning_rate": 4.572727272727273e-06, "num_tokens": 4035226.0, "completions/mean_length": 68.125, "completions/min_length": 68.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.125, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9975280165672302, "rewards/meter/std": 4.185540819889866e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9975280165672302, "rewards/total_composite/std": 4.185540819889866e-05, "reward": 0.9975280165672302, "reward_std": 4.18471208831761e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0064887660555541515, "sampling/sampling_logp_difference/max": 0.6608014106750488, "sampling/importance_sampling_ratio/min": 0.516437292098999, "sampling/importance_sampling_ratio/mean": 1.0018044710159302, "sampling/importance_sampling_ratio/max": 1.4639054536819458, "entropy": 0.04893740452826023, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9975280165672302, "reward_meter_mean": 0.9975280165672302, "reward_meter_std": 4.185540819889866e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9975280165672302, "reward_total_composite_std": 4.185540819889866e-05} {"timestamp_utc": "2026-04-12T00:51:25Z", "mode": "train", "global_step": 1793, "epoch": 0.07201670884042254, "loss": 0.0024, "grad_norm": 2.3365345001220703, "learning_rate": 4.56969696969697e-06, "num_tokens": 4037239.0, "completions/mean_length": 94.625, "completions/min_length": 94.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.625, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.9981467723846436, "rewards/meter/std": 4.623903805622831e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.8733782768249512, "rewards/total_composite/std": 0.1033172756433487, "reward": 0.8733782768249512, "reward_std": 0.10331728309392929, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003089956007897854, "sampling/sampling_logp_difference/max": 0.6266226768493652, "sampling/importance_sampling_ratio/min": 0.5343936085700989, "sampling/importance_sampling_ratio/mean": 1.0002309083938599, "sampling/importance_sampling_ratio/max": 1.2057961225509644, "entropy": 0.021367451641708612, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0013157895300537348, "clip_ratio/high_max": 0.0013157895300537348, "clip_ratio/region_mean": 0.0013157895300537348, "reward_total_mean": 0.8733782768249512, "reward_meter_mean": 0.9981467723846436, "reward_meter_std": 4.623903805622831e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.8733782768249512, "reward_total_composite_std": 0.1033172756433487} {"timestamp_utc": "2026-04-12T00:51:31Z", "mode": "train", "global_step": 1794, "epoch": 0.07205687432220749, "loss": -0.0075, "grad_norm": 1.2437056303024292, "learning_rate": 4.566666666666667e-06, "num_tokens": 4040774.0, "completions/mean_length": 205.875, "completions/min_length": 204.0, "completions/max_length": 213.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 205.875, "completions/min_terminated_length": 204.0, "completions/max_terminated_length": 213.0, "rewards/meter/mean": 0.9908320307731628, "rewards/meter/std": 0.000867572205606848, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6607142686843872, "rewards/repeat_penalty/std": 0.06331465393304825, "rewards/total_composite/mean": 0.6546941995620728, "rewards/total_composite/std": 0.06324289739131927, "reward": 0.6546941995620728, "reward_std": 0.06324288249015808, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009390618652105331, "sampling/sampling_logp_difference/max": 1.3305549621582031, "sampling/importance_sampling_ratio/min": 0.26433053612709045, "sampling/importance_sampling_ratio/mean": 1.0008214712142944, "sampling/importance_sampling_ratio/max": 1.7237244844436646, "entropy": 0.0551890037022531, "clip_ratio/low_mean": 0.0030517695122398436, "clip_ratio/low_min": 0.0030517695122398436, "clip_ratio/high_mean": 0.002393221133388579, "clip_ratio/high_max": 0.002393221133388579, "clip_ratio/region_mean": 0.0054449906456284225, "reward_total_mean": 0.6546941995620728, "reward_meter_mean": 0.9908320307731628, "reward_meter_std": 0.000867572205606848, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6607142686843872, "reward_repeat_penalty_std": 0.06331465393304825, "reward_total_composite_mean": 0.6546941995620728, "reward_total_composite_std": 0.06324289739131927} {"timestamp_utc": "2026-04-12T00:51:37Z", "mode": "train", "global_step": 1795, "epoch": 0.07209703980399244, "loss": 0.0056, "grad_norm": 2.7669808864593506, "learning_rate": 4.563636363636364e-06, "num_tokens": 4043010.0, "completions/mean_length": 118.5, "completions/min_length": 116.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.5, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.9971722364425659, "rewards/meter/std": 0.0011399141512811184, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9722900390625, "rewards/total_composite/std": 0.0711437240242958, "reward": 0.9722900390625, "reward_std": 0.07114371657371521, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04540203511714935, "sampling/sampling_logp_difference/max": 1.2590808868408203, "sampling/importance_sampling_ratio/min": 0.28391486406326294, "sampling/importance_sampling_ratio/mean": 1.014491319656372, "sampling/importance_sampling_ratio/max": 1.7191333770751953, "entropy": 0.46814702078700066, "clip_ratio/low_mean": 0.004201680887490511, "clip_ratio/low_min": 0.004201680887490511, "clip_ratio/high_mean": 0.030286709661595523, "clip_ratio/high_max": 0.030286709661595523, "clip_ratio/region_mean": 0.034488390549086034, "reward_total_mean": 0.9722900390625, "reward_meter_mean": 0.9971722364425659, "reward_meter_std": 0.0011399141512811184, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9722900390625, "reward_total_composite_std": 0.0711437240242958} {"timestamp_utc": "2026-04-12T00:51:43Z", "mode": "train", "global_step": 1796, "epoch": 0.0721372052857774, "loss": 0.0069, "grad_norm": 2.253188371658325, "learning_rate": 4.560606060606061e-06, "num_tokens": 4045960.0, "completions/mean_length": 184.75, "completions/min_length": 168.0, "completions/max_length": 199.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 184.75, "completions/min_terminated_length": 168.0, "completions/max_terminated_length": 199.0, "rewards/meter/mean": 0.9870539903640747, "rewards/meter/std": 0.024071279913187027, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.8888888955116272, "rewards/repeat_penalty/std": 0.10286889225244522, "rewards/total_composite/mean": 0.7317129969596863, "rewards/total_composite/std": 0.30691879987716675, "reward": 0.7317129969596863, "reward_std": 0.30691882967948914, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05264398828148842, "sampling/sampling_logp_difference/max": 1.2949190139770508, "sampling/importance_sampling_ratio/min": 0.27392005920410156, "sampling/importance_sampling_ratio/mean": 1.0125421285629272, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5114907324314117, "clip_ratio/low_mean": 0.002659574383869767, "clip_ratio/low_min": 0.002659574383869767, "clip_ratio/high_mean": 0.024542337749153376, "clip_ratio/high_max": 0.024542337749153376, "clip_ratio/region_mean": 0.027201912133023143, "reward_total_mean": 0.7317129969596863, "reward_meter_mean": 0.9870539903640747, "reward_meter_std": 0.024071279913187027, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.8888888955116272, "reward_repeat_penalty_std": 0.10286889225244522, "reward_total_composite_mean": 0.7317129969596863, "reward_total_composite_std": 0.30691879987716675} {"timestamp_utc": "2026-04-12T00:51:52Z", "mode": "train", "global_step": 1797, "epoch": 0.07217737076756235, "loss": 0.0079, "grad_norm": 1.9448902606964111, "learning_rate": 4.557575757575758e-06, "num_tokens": 4051243.0, "completions/mean_length": 420.375, "completions/min_length": 395.0, "completions/max_length": 458.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 420.375, "completions/min_terminated_length": 395.0, "completions/max_terminated_length": 458.0, "rewards/meter/mean": 0.448710560798645, "rewards/meter/std": 0.4174151122570038, "rewards/count_adherence/mean": 0.5625, "rewards/count_adherence/std": 0.023145509883761406, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7848790884017944, "rewards/repeat_penalty/std": 0.13375473022460938, "rewards/total_composite/mean": 0.2035287767648697, "rewards/total_composite/std": 0.17783083021640778, "reward": 0.2035287767648697, "reward_std": 0.17783081531524658, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.036439694464206696, "sampling/sampling_logp_difference/max": 7.995334625244141, "sampling/importance_sampling_ratio/min": 0.0003370313497725874, "sampling/importance_sampling_ratio/mean": 1.0021564960479736, "sampling/importance_sampling_ratio/max": 1.8132941722869873, "entropy": 0.22779023833572865, "clip_ratio/low_mean": 0.010334618214983493, "clip_ratio/low_min": 0.010334618214983493, "clip_ratio/high_mean": 0.01244131033308804, "clip_ratio/high_max": 0.01244131033308804, "clip_ratio/region_mean": 0.022775928548071533, "reward_total_mean": 0.2035287767648697, "reward_meter_mean": 0.448710560798645, "reward_meter_std": 0.4174151122570038, "reward_count_adherence_mean": 0.5625, "reward_count_adherence_std": 0.023145509883761406, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7848790884017944, "reward_repeat_penalty_std": 0.13375473022460938, "reward_total_composite_mean": 0.2035287767648697, "reward_total_composite_std": 0.17783083021640778} {"timestamp_utc": "2026-04-12T00:51:59Z", "mode": "train", "global_step": 1798, "epoch": 0.0722175362493473, "loss": 0.0113, "grad_norm": 2.2561349868774414, "learning_rate": 4.554545454545455e-06, "num_tokens": 4054231.0, "completions/mean_length": 182.5, "completions/min_length": 172.0, "completions/max_length": 188.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 182.5, "completions/min_terminated_length": 172.0, "completions/max_terminated_length": 188.0, "rewards/meter/mean": 0.9977061152458191, "rewards/meter/std": 0.001272720517590642, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9305555820465088, "rewards/repeat_penalty/std": 0.08266931027173996, "rewards/total_composite/mean": 0.9284070730209351, "rewards/total_composite/std": 0.0822766125202179, "reward": 0.9284070730209351, "reward_std": 0.08227662742137909, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03487614914774895, "sampling/sampling_logp_difference/max": 1.8383588790893555, "sampling/importance_sampling_ratio/min": 0.15907828509807587, "sampling/importance_sampling_ratio/mean": 1.008242130279541, "sampling/importance_sampling_ratio/max": 1.7942215204238892, "entropy": 0.3780089020729065, "clip_ratio/low_mean": 0.013545665598940104, "clip_ratio/low_min": 0.013545665598940104, "clip_ratio/high_mean": 0.017403612146154046, "clip_ratio/high_max": 0.017403612146154046, "clip_ratio/region_mean": 0.03094927774509415, "reward_total_mean": 0.9284070730209351, "reward_meter_mean": 0.9977061152458191, "reward_meter_std": 0.001272720517590642, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9305555820465088, "reward_repeat_penalty_std": 0.08266931027173996, "reward_total_composite_mean": 0.9284070730209351, "reward_total_composite_std": 0.0822766125202179} {"timestamp_utc": "2026-04-12T00:52:04Z", "mode": "train", "global_step": 1799, "epoch": 0.07225770173113226, "loss": -0.0358, "grad_norm": 5.813068866729736, "learning_rate": 4.551515151515152e-06, "num_tokens": 4056037.0, "completions/mean_length": 68.75, "completions/min_length": 63.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.75, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.7557946443557739, "rewards/meter/std": 0.27249956130981445, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7557946443557739, "rewards/total_composite/std": 0.27249956130981445, "reward": 0.7557946443557739, "reward_std": 0.27249956130981445, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.054108865559101105, "sampling/sampling_logp_difference/max": 0.9117650985717773, "sampling/importance_sampling_ratio/min": 0.40181437134742737, "sampling/importance_sampling_ratio/mean": 1.0174649953842163, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5148973725736141, "clip_ratio/low_mean": 0.018777355086058378, "clip_ratio/low_min": 0.018777355086058378, "clip_ratio/high_mean": 0.019246643991209567, "clip_ratio/high_max": 0.019246643991209567, "clip_ratio/region_mean": 0.038023999077267945, "reward_total_mean": 0.7557946443557739, "reward_meter_mean": 0.7557946443557739, "reward_meter_std": 0.27249956130981445, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7557946443557739, "reward_total_composite_std": 0.27249956130981445} {"timestamp_utc": "2026-04-12T00:52:08Z", "mode": "train", "global_step": 1800, "epoch": 0.07229786721291721, "loss": 0.0696, "grad_norm": 13.604365348815918, "learning_rate": 4.548484848484849e-06, "num_tokens": 4057576.0, "completions/mean_length": 40.375, "completions/min_length": 39.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.375, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.9156763553619385, "rewards/meter/std": 0.20896102488040924, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9156763553619385, "rewards/total_composite/std": 0.20896102488040924, "reward": 0.9156763553619385, "reward_std": 0.20896100997924805, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07238361239433289, "sampling/sampling_logp_difference/max": 1.5261836051940918, "sampling/importance_sampling_ratio/min": 0.217363640666008, "sampling/importance_sampling_ratio/mean": 1.0134040117263794, "sampling/importance_sampling_ratio/max": 1.8849577903747559, "entropy": 0.5478580147027969, "clip_ratio/low_mean": 0.01666666753590107, "clip_ratio/low_min": 0.01666666753590107, "clip_ratio/high_mean": 0.05649038520641625, "clip_ratio/high_max": 0.05649038520641625, "clip_ratio/region_mean": 0.07315705274231732, "reward_total_mean": 0.9156763553619385, "reward_meter_mean": 0.9156763553619385, "reward_meter_std": 0.20896102488040924, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9156763553619385, "reward_total_composite_std": 0.20896102488040924} {"timestamp_utc": "2026-04-12T00:53:14Z", "mode": "eval", "global_step": 1800, "epoch": 0.07229786721291721, "eval_loss": NaN, "eval_runtime": 66.0173, "eval_samples_per_second": 1.575, "eval_steps_per_second": 0.197, "eval_num_tokens": 4057576.0, "eval_completions/mean_length": 192.78846153846155, "eval_completions/min_length": 63.84615384615385, "eval_completions/max_length": 341.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 192.78846153846155, "eval_completions/min_terminated_length": 63.84615384615385, "eval_completions/max_terminated_length": 341.0, "eval_rewards/meter/mean": 0.6883980471354264, "eval_rewards/meter/std": 0.388204580746018, "eval_rewards/count_adherence/mean": 0.8753787233279302, "eval_rewards/count_adherence/std": 0.13936235010623932, "eval_rewards/arabic_clean/mean": 0.9807692307692307, "eval_rewards/arabic_clean/std": 0.05439282839114849, "eval_rewards/repeat_penalty/mean": 0.8605746856102576, "eval_rewards/repeat_penalty/std": 0.13569565231983477, "eval_rewards/total_composite/mean": 0.5133193800082574, "eval_rewards/total_composite/std": 0.346407216328841, "eval_reward": 0.5133193800082574, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.02931356322593414, "eval_sampling/sampling_logp_difference/max": 1.0275482031015248, "eval_sampling/importance_sampling_ratio/min": 0.36259872638262236, "eval_sampling/importance_sampling_ratio/mean": 1.0084307743952825, "eval_sampling/importance_sampling_ratio/max": 1.5431275459436269, "eval_entropy": 0.3313331672778496, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5133193800082574, "eval_reward_meter_mean": 0.6883980471354264, "eval_reward_meter_std": 0.388204580746018, "eval_reward_count_adherence_mean": 0.8753787233279302, "eval_reward_count_adherence_std": 0.13936235010623932, "eval_reward_arabic_clean_mean": 0.9807692307692307, "eval_reward_arabic_clean_std": 0.05439282839114849, "eval_reward_repeat_penalty_mean": 0.8605746856102576, "eval_reward_repeat_penalty_std": 0.13569565231983477, "eval_reward_total_composite_mean": 0.5133193800082574, "eval_reward_total_composite_std": 0.346407216328841} {"timestamp_utc": "2026-04-12T00:53:23Z", "mode": "train", "global_step": 1801, "epoch": 0.07233803269470217, "loss": 0.0329, "grad_norm": 4.170017719268799, "learning_rate": 4.5454545454545455e-06, "num_tokens": 4061109.0, "completions/mean_length": 241.625, "completions/min_length": 227.0, "completions/max_length": 254.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 241.625, "completions/min_terminated_length": 227.0, "completions/max_terminated_length": 254.0, "rewards/meter/mean": 0.5351053476333618, "rewards/meter/std": 0.38701698184013367, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.9261363744735718, "rewards/repeat_penalty/std": 0.05368336662650108, "rewards/total_composite/mean": 0.39037013053894043, "rewards/total_composite/std": 0.3217414319515228, "reward": 0.39037013053894043, "reward_std": 0.3217414319515228, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0780494287610054, "sampling/sampling_logp_difference/max": 2.623555898666382, "sampling/importance_sampling_ratio/min": 0.07254444062709808, "sampling/importance_sampling_ratio/mean": 1.0214029550552368, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6957518979907036, "clip_ratio/low_mean": 0.02372721955180168, "clip_ratio/low_min": 0.02372721955180168, "clip_ratio/high_mean": 0.027773853624239564, "clip_ratio/high_max": 0.027773853624239564, "clip_ratio/region_mean": 0.051501073176041245, "reward_total_mean": 0.39037013053894043, "reward_meter_mean": 0.5351053476333618, "reward_meter_std": 0.38701698184013367, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.9261363744735718, "reward_repeat_penalty_std": 0.05368336662650108, "reward_total_composite_mean": 0.39037013053894043, "reward_total_composite_std": 0.3217414319515228} {"timestamp_utc": "2026-04-12T00:53:29Z", "mode": "train", "global_step": 1802, "epoch": 0.07237819817648712, "loss": 0.038, "grad_norm": 4.134006500244141, "learning_rate": 4.542424242424243e-06, "num_tokens": 4063157.0, "completions/mean_length": 103.0, "completions/min_length": 98.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.0, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.9173542261123657, "rewards/meter/std": 0.045704178512096405, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.8030074238777161, "rewards/total_composite/std": 0.10497334599494934, "reward": 0.8030074238777161, "reward_std": 0.10497336089611053, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.036168623715639114, "sampling/sampling_logp_difference/max": 1.6687889099121094, "sampling/importance_sampling_ratio/min": 0.18847519159317017, "sampling/importance_sampling_ratio/mean": 1.0087733268737793, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3216414675116539, "clip_ratio/low_mean": 0.017737353453412652, "clip_ratio/low_min": 0.017737353453412652, "clip_ratio/high_mean": 0.01266491471324116, "clip_ratio/high_max": 0.01266491471324116, "clip_ratio/region_mean": 0.030402268166653812, "reward_total_mean": 0.8030074238777161, "reward_meter_mean": 0.9173542261123657, "reward_meter_std": 0.045704178512096405, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.8030074238777161, "reward_total_composite_std": 0.10497334599494934} {"timestamp_utc": "2026-04-12T00:53:33Z", "mode": "train", "global_step": 1803, "epoch": 0.07241836365827208, "loss": -0.0189, "grad_norm": 7.020390510559082, "learning_rate": 4.539393939393939e-06, "num_tokens": 4064817.0, "completions/mean_length": 34.5, "completions/min_length": 33.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.5, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9750309586524963, "rewards/meter/std": 0.01676092855632305, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9750309586524963, "rewards/total_composite/std": 0.01676092855632305, "reward": 0.9750309586524963, "reward_std": 0.01676092855632305, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06769955903291702, "sampling/sampling_logp_difference/max": 1.4559955596923828, "sampling/importance_sampling_ratio/min": 0.2331681251525879, "sampling/importance_sampling_ratio/mean": 1.008634090423584, "sampling/importance_sampling_ratio/max": 1.694462537765503, "entropy": 0.6431399546563625, "clip_ratio/low_mean": 0.022176598431542516, "clip_ratio/low_min": 0.022176598431542516, "clip_ratio/high_mean": 0.02186147216707468, "clip_ratio/high_max": 0.02186147216707468, "clip_ratio/region_mean": 0.044038070598617196, "reward_total_mean": 0.9750309586524963, "reward_meter_mean": 0.9750309586524963, "reward_meter_std": 0.01676092855632305, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9750309586524963, "reward_total_composite_std": 0.01676092855632305} {"timestamp_utc": "2026-04-12T00:53:38Z", "mode": "train", "global_step": 1804, "epoch": 0.07245852914005703, "loss": 0.0014, "grad_norm": 2.0322115421295166, "learning_rate": 4.5363636363636364e-06, "num_tokens": 4066459.0, "completions/mean_length": 57.25, "completions/min_length": 57.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.25, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9932329654693604, "rewards/meter/std": 0.0008318190230056643, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9932329654693604, "rewards/total_composite/std": 0.0008318190230056643, "reward": 0.9932329654693604, "reward_std": 0.0008318254840560257, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026665696874260902, "sampling/sampling_logp_difference/max": 1.751237392425537, "sampling/importance_sampling_ratio/min": 0.4126149117946625, "sampling/importance_sampling_ratio/mean": 1.0055011510849, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.13894780911505222, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.010927102295681834, "clip_ratio/high_max": 0.010927102295681834, "clip_ratio/region_mean": 0.010927102295681834, "reward_total_mean": 0.9932329654693604, "reward_meter_mean": 0.9932329654693604, "reward_meter_std": 0.0008318190230056643, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9932329654693604, "reward_total_composite_std": 0.0008318190230056643} {"timestamp_utc": "2026-04-12T00:53:42Z", "mode": "train", "global_step": 1805, "epoch": 0.07249869462184198, "loss": -0.0002, "grad_norm": 2.5561840534210205, "learning_rate": 4.533333333333334e-06, "num_tokens": 4068131.0, "completions/mean_length": 63.0, "completions/min_length": 63.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.0, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9917310476303101, "rewards/meter/std": 0.014378287829458714, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9917310476303101, "rewards/total_composite/std": 0.014378287829458714, "reward": 0.9917310476303101, "reward_std": 0.014378308318555355, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010517938062548637, "sampling/sampling_logp_difference/max": 0.489483118057251, "sampling/importance_sampling_ratio/min": 0.612943172454834, "sampling/importance_sampling_ratio/mean": 1.0059313774108887, "sampling/importance_sampling_ratio/max": 1.4607477188110352, "entropy": 0.09062223881483078, "clip_ratio/low_mean": 0.0019841270986944437, "clip_ratio/low_min": 0.0019841270986944437, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0019841270986944437, "reward_total_mean": 0.9917310476303101, "reward_meter_mean": 0.9917310476303101, "reward_meter_std": 0.014378287829458714, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9917310476303101, "reward_total_composite_std": 0.014378287829458714} {"timestamp_utc": "2026-04-12T00:53:47Z", "mode": "train", "global_step": 1806, "epoch": 0.07253886010362694, "loss": 0.0586, "grad_norm": 5.9204182624816895, "learning_rate": 4.53030303030303e-06, "num_tokens": 4069918.0, "completions/mean_length": 80.375, "completions/min_length": 74.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.375, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.774915874004364, "rewards/meter/std": 0.3431837856769562, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6509301066398621, "rewards/total_composite/std": 0.42339858412742615, "reward": 0.6509301066398621, "reward_std": 0.42339858412742615, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09225737303495407, "sampling/sampling_logp_difference/max": 3.772925615310669, "sampling/importance_sampling_ratio/min": 0.02298472262918949, "sampling/importance_sampling_ratio/mean": 1.022961139678955, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9013412185013294, "clip_ratio/low_mean": 0.013943745056167245, "clip_ratio/low_min": 0.013943745056167245, "clip_ratio/high_mean": 0.03447378391865641, "clip_ratio/high_max": 0.03447378391865641, "clip_ratio/region_mean": 0.048417528974823654, "reward_total_mean": 0.6509301066398621, "reward_meter_mean": 0.774915874004364, "reward_meter_std": 0.3431837856769562, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6509301066398621, "reward_total_composite_std": 0.42339858412742615} {"timestamp_utc": "2026-04-12T00:53:51Z", "mode": "train", "global_step": 1807, "epoch": 0.07257902558541189, "loss": 0.0059, "grad_norm": 3.1104016304016113, "learning_rate": 4.527272727272727e-06, "num_tokens": 4071641.0, "completions/mean_length": 57.375, "completions/min_length": 57.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.375, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9939553737640381, "rewards/meter/std": 0.0008149455534294248, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9939553737640381, "rewards/total_composite/std": 0.0008149455534294248, "reward": 0.9939553737640381, "reward_std": 0.0008149410132318735, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021058987826108932, "sampling/sampling_logp_difference/max": 0.8907175064086914, "sampling/importance_sampling_ratio/min": 0.41036123037338257, "sampling/importance_sampling_ratio/mean": 1.0032254457473755, "sampling/importance_sampling_ratio/max": 1.5543711185455322, "entropy": 0.13474871311336756, "clip_ratio/low_mean": 0.008621971355751157, "clip_ratio/low_min": 0.008621971355751157, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.008621971355751157, "reward_total_mean": 0.9939553737640381, "reward_meter_mean": 0.9939553737640381, "reward_meter_std": 0.0008149455534294248, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9939553737640381, "reward_total_composite_std": 0.0008149455534294248} {"timestamp_utc": "2026-04-12T00:53:57Z", "mode": "train", "global_step": 1808, "epoch": 0.07261919106719684, "loss": -0.0419, "grad_norm": 4.073712348937988, "learning_rate": 4.524242424242425e-06, "num_tokens": 4074374.0, "completions/mean_length": 157.625, "completions/min_length": 132.0, "completions/max_length": 169.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 157.625, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 169.0, "rewards/meter/mean": 0.8494125604629517, "rewards/meter/std": 0.2693995535373688, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7510942220687866, "rewards/total_composite/std": 0.41580459475517273, "reward": 0.7510942220687866, "reward_std": 0.41580459475517273, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0892598107457161, "sampling/sampling_logp_difference/max": 1.8835735321044922, "sampling/importance_sampling_ratio/min": 0.15204578638076782, "sampling/importance_sampling_ratio/mean": 1.0168873071670532, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9014198370277882, "clip_ratio/low_mean": 0.01045380299910903, "clip_ratio/low_min": 0.01045380299910903, "clip_ratio/high_mean": 0.039301327895373106, "clip_ratio/high_max": 0.039301327895373106, "clip_ratio/region_mean": 0.049755130894482136, "reward_total_mean": 0.7510942220687866, "reward_meter_mean": 0.8494125604629517, "reward_meter_std": 0.2693995535373688, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7510942220687866, "reward_total_composite_std": 0.41580459475517273} {"timestamp_utc": "2026-04-12T00:54:03Z", "mode": "train", "global_step": 1809, "epoch": 0.0726593565489818, "loss": 0.0062, "grad_norm": 4.39199161529541, "learning_rate": 4.521212121212122e-06, "num_tokens": 4077130.0, "completions/mean_length": 160.5, "completions/min_length": 152.0, "completions/max_length": 172.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 160.5, "completions/min_terminated_length": 152.0, "completions/max_terminated_length": 172.0, "rewards/meter/mean": 0.9320476055145264, "rewards/meter/std": 0.11087671667337418, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9022413492202759, "rewards/total_composite/std": 0.13386203348636627, "reward": 0.9022413492202759, "reward_std": 0.13386201858520508, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10231172293424606, "sampling/sampling_logp_difference/max": 1.5400352478027344, "sampling/importance_sampling_ratio/min": 0.21437355875968933, "sampling/importance_sampling_ratio/mean": 1.0246384143829346, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1915706768631935, "clip_ratio/low_mean": 0.017330652568489313, "clip_ratio/low_min": 0.017330652568489313, "clip_ratio/high_mean": 0.044713106006383896, "clip_ratio/high_max": 0.044713106006383896, "clip_ratio/region_mean": 0.06204375857487321, "reward_total_mean": 0.9022413492202759, "reward_meter_mean": 0.9320476055145264, "reward_meter_std": 0.11087671667337418, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9022413492202759, "reward_total_composite_std": 0.13386203348636627} {"timestamp_utc": "2026-04-12T00:54:08Z", "mode": "train", "global_step": 1810, "epoch": 0.07269952203076675, "loss": -0.0001, "grad_norm": 2.1702988147735596, "learning_rate": 4.518181818181819e-06, "num_tokens": 4079618.0, "completions/mean_length": 136.0, "completions/min_length": 132.0, "completions/max_length": 144.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 136.0, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 144.0, "rewards/meter/mean": 0.8782405853271484, "rewards/meter/std": 0.031040111556649208, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8035714030265808, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.7060449123382568, "rewards/total_composite/std": 0.07351230084896088, "reward": 0.7060449123382568, "reward_std": 0.07351230829954147, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022378606721758842, "sampling/sampling_logp_difference/max": 1.1933460235595703, "sampling/importance_sampling_ratio/min": 0.30320504307746887, "sampling/importance_sampling_ratio/mean": 1.0071299076080322, "sampling/importance_sampling_ratio/max": 1.6531912088394165, "entropy": 0.24091620557010174, "clip_ratio/low_mean": 0.006453010253608227, "clip_ratio/low_min": 0.006453010253608227, "clip_ratio/high_mean": 0.012699689657893032, "clip_ratio/high_max": 0.012699689657893032, "clip_ratio/region_mean": 0.01915269991150126, "reward_total_mean": 0.7060449123382568, "reward_meter_mean": 0.8782405853271484, "reward_meter_std": 0.031040111556649208, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8035714030265808, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.7060449123382568, "reward_total_composite_std": 0.07351230084896088} {"timestamp_utc": "2026-04-12T00:54:13Z", "mode": "train", "global_step": 1811, "epoch": 0.0727396875125517, "loss": 0.0067, "grad_norm": 4.185992240905762, "learning_rate": 4.5151515151515155e-06, "num_tokens": 4081773.0, "completions/mean_length": 104.375, "completions/min_length": 99.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 104.375, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.8305845260620117, "rewards/meter/std": 0.25914400815963745, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8305845260620117, "rewards/total_composite/std": 0.25914400815963745, "reward": 0.8305845260620117, "reward_std": 0.25914397835731506, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.058025941252708435, "sampling/sampling_logp_difference/max": 1.5509662628173828, "sampling/importance_sampling_ratio/min": 0.21204298734664917, "sampling/importance_sampling_ratio/mean": 1.0125762224197388, "sampling/importance_sampling_ratio/max": 1.7020655870437622, "entropy": 0.5101095288991928, "clip_ratio/low_mean": 0.006021729204803705, "clip_ratio/low_min": 0.006021729204803705, "clip_ratio/high_mean": 0.037365528754889965, "clip_ratio/high_max": 0.037365528754889965, "clip_ratio/region_mean": 0.04338725795969367, "reward_total_mean": 0.8305845260620117, "reward_meter_mean": 0.8305845260620117, "reward_meter_std": 0.25914400815963745, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8305845260620117, "reward_total_composite_std": 0.25914400815963745} {"timestamp_utc": "2026-04-12T00:54:18Z", "mode": "train", "global_step": 1812, "epoch": 0.07277985299433666, "loss": 0.0156, "grad_norm": 7.030357360839844, "learning_rate": 4.512121212121213e-06, "num_tokens": 4083458.0, "completions/mean_length": 62.625, "completions/min_length": 59.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.625, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.7619152069091797, "rewards/meter/std": 0.4342895746231079, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7619152069091797, "rewards/total_composite/std": 0.4342895746231079, "reward": 0.7619152069091797, "reward_std": 0.4342895746231079, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024434061720967293, "sampling/sampling_logp_difference/max": 1.121795654296875, "sampling/importance_sampling_ratio/min": 0.3256944417953491, "sampling/importance_sampling_ratio/mean": 1.0049937963485718, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.13391203712671995, "clip_ratio/low_mean": 0.003937252098694444, "clip_ratio/low_min": 0.003937252098694444, "clip_ratio/high_mean": 0.009920635493472219, "clip_ratio/high_max": 0.009920635493472219, "clip_ratio/region_mean": 0.013857887592166662, "reward_total_mean": 0.7619152069091797, "reward_meter_mean": 0.7619152069091797, "reward_meter_std": 0.4342895746231079, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7619152069091797, "reward_total_composite_std": 0.4342895746231079} {"timestamp_utc": "2026-04-12T00:54:24Z", "mode": "train", "global_step": 1813, "epoch": 0.07282001847612161, "loss": -0.0181, "grad_norm": 3.172938346862793, "learning_rate": 4.50909090909091e-06, "num_tokens": 4086534.0, "completions/mean_length": 176.5, "completions/min_length": 160.0, "completions/max_length": 195.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 176.5, "completions/min_terminated_length": 160.0, "completions/max_terminated_length": 195.0, "rewards/meter/mean": 0.9555693864822388, "rewards/meter/std": 0.09552352130413055, "rewards/count_adherence/mean": 0.8500000238418579, "rewards/count_adherence/std": 0.09258200973272324, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8140916228294373, "rewards/total_composite/std": 0.132888063788414, "reward": 0.8140916228294373, "reward_std": 0.132888063788414, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07848730683326721, "sampling/sampling_logp_difference/max": 1.3485078811645508, "sampling/importance_sampling_ratio/min": 0.2596273720264435, "sampling/importance_sampling_ratio/mean": 1.0195351839065552, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7889967784285545, "clip_ratio/low_mean": 0.030510480049997568, "clip_ratio/low_min": 0.030510480049997568, "clip_ratio/high_mean": 0.015669515822082758, "clip_ratio/high_max": 0.015669515822082758, "clip_ratio/region_mean": 0.046179995872080326, "reward_total_mean": 0.8140916228294373, "reward_meter_mean": 0.9555693864822388, "reward_meter_std": 0.09552352130413055, "reward_count_adherence_mean": 0.8500000238418579, "reward_count_adherence_std": 0.09258200973272324, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8140916228294373, "reward_total_composite_std": 0.132888063788414} {"timestamp_utc": "2026-04-12T00:54:29Z", "mode": "train", "global_step": 1814, "epoch": 0.07286018395790657, "loss": -0.0068, "grad_norm": 2.741095781326294, "learning_rate": 4.5060606060606065e-06, "num_tokens": 4088604.0, "completions/mean_length": 95.75, "completions/min_length": 93.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.75, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9976317882537842, "rewards/meter/std": 0.0016386967618018389, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9976317882537842, "rewards/total_composite/std": 0.0016386967618018389, "reward": 0.9976317882537842, "reward_std": 0.0016386950155720115, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006327650509774685, "sampling/sampling_logp_difference/max": 0.28154802322387695, "sampling/importance_sampling_ratio/min": 0.8114128112792969, "sampling/importance_sampling_ratio/mean": 1.0037214756011963, "sampling/importance_sampling_ratio/max": 1.3251795768737793, "entropy": 0.04789540730416775, "clip_ratio/low_mean": 0.0013440860202535987, "clip_ratio/low_min": 0.0013440860202535987, "clip_ratio/high_mean": 0.0025775935500860214, "clip_ratio/high_max": 0.0025775935500860214, "clip_ratio/region_mean": 0.00392167957033962, "reward_total_mean": 0.9976317882537842, "reward_meter_mean": 0.9976317882537842, "reward_meter_std": 0.0016386967618018389, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9976317882537842, "reward_total_composite_std": 0.0016386967618018389} {"timestamp_utc": "2026-04-12T00:54:35Z", "mode": "train", "global_step": 1815, "epoch": 0.07290034943969152, "loss": 0.0014, "grad_norm": 2.079590320587158, "learning_rate": 4.503030303030304e-06, "num_tokens": 4091580.0, "completions/mean_length": 175.0, "completions/min_length": 163.0, "completions/max_length": 185.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 175.0, "completions/min_terminated_length": 163.0, "completions/max_terminated_length": 185.0, "rewards/meter/mean": 0.9942752122879028, "rewards/meter/std": 0.0037329280748963356, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8152778148651123, "rewards/repeat_penalty/std": 0.05002203211188316, "rewards/total_composite/mean": 0.6754500865936279, "rewards/total_composite/std": 0.04033276066184044, "reward": 0.6754500865936279, "reward_std": 0.04033275321125984, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019572407007217407, "sampling/sampling_logp_difference/max": 2.9577598571777344, "sampling/importance_sampling_ratio/min": 0.05193512886762619, "sampling/importance_sampling_ratio/mean": 1.0022673606872559, "sampling/importance_sampling_ratio/max": 1.8784008026123047, "entropy": 0.12415077816694975, "clip_ratio/low_mean": 0.005773193319328129, "clip_ratio/low_min": 0.005773193319328129, "clip_ratio/high_mean": 0.004235348082147539, "clip_ratio/high_max": 0.004235348082147539, "clip_ratio/region_mean": 0.010008541401475668, "reward_total_mean": 0.6754500865936279, "reward_meter_mean": 0.9942752122879028, "reward_meter_std": 0.0037329280748963356, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8152778148651123, "reward_repeat_penalty_std": 0.05002203211188316, "reward_total_composite_mean": 0.6754500865936279, "reward_total_composite_std": 0.04033276066184044} {"timestamp_utc": "2026-04-12T00:54:40Z", "mode": "train", "global_step": 1816, "epoch": 0.07294051492147648, "loss": 0.3243, "grad_norm": 7.065499782562256, "learning_rate": 4.5e-06, "num_tokens": 4093095.0, "completions/mean_length": 51.375, "completions/min_length": 36.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.375, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9962137937545776, "rewards/meter/std": 0.0032140789553523064, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.5175492167472839, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6230374574661255, "rewards/total_composite/std": 0.5159306526184082, "reward": 0.6230374574661255, "reward_std": 0.5159306526184082, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08005251735448837, "sampling/sampling_logp_difference/max": 1.0488758087158203, "sampling/importance_sampling_ratio/min": 0.3503313660621643, "sampling/importance_sampling_ratio/mean": 1.0141249895095825, "sampling/importance_sampling_ratio/max": 1.9115855693817139, "entropy": 0.709318995475769, "clip_ratio/low_mean": 0.022342659067362547, "clip_ratio/low_min": 0.022342659067362547, "clip_ratio/high_mean": 0.03601037664338946, "clip_ratio/high_max": 0.03601037664338946, "clip_ratio/region_mean": 0.05835303571075201, "reward_total_mean": 0.6230374574661255, "reward_meter_mean": 0.9962137937545776, "reward_meter_std": 0.0032140789553523064, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.5175492167472839, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6230374574661255, "reward_total_composite_std": 0.5159306526184082} {"timestamp_utc": "2026-04-12T00:54:44Z", "mode": "train", "global_step": 1817, "epoch": 0.07298068040326143, "loss": 0.0293, "grad_norm": 9.501838684082031, "learning_rate": 4.496969696969697e-06, "num_tokens": 4094892.0, "completions/mean_length": 72.625, "completions/min_length": 68.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.625, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.8915443420410156, "rewards/meter/std": 0.21483440697193146, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8915443420410156, "rewards/total_composite/std": 0.21483440697193146, "reward": 0.8915443420410156, "reward_std": 0.21483437716960907, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08924838155508041, "sampling/sampling_logp_difference/max": 2.1535000801086426, "sampling/importance_sampling_ratio/min": 0.11607716232538223, "sampling/importance_sampling_ratio/mean": 1.027990460395813, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8766130320727825, "clip_ratio/low_mean": 0.011928289197385311, "clip_ratio/low_min": 0.011928289197385311, "clip_ratio/high_mean": 0.050623598508536816, "clip_ratio/high_max": 0.050623598508536816, "clip_ratio/region_mean": 0.06255188770592213, "reward_total_mean": 0.8915443420410156, "reward_meter_mean": 0.8915443420410156, "reward_meter_std": 0.21483440697193146, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8915443420410156, "reward_total_composite_std": 0.21483440697193146} {"timestamp_utc": "2026-04-12T00:54:49Z", "mode": "train", "global_step": 1818, "epoch": 0.07302084588504638, "loss": -0.0031, "grad_norm": 0.42083650827407837, "learning_rate": 4.493939393939395e-06, "num_tokens": 4097204.0, "completions/mean_length": 102.0, "completions/min_length": 100.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.0, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.9974691271781921, "rewards/meter/std": 6.317263614619151e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9974691271781921, "rewards/total_composite/std": 6.317263614619151e-05, "reward": 0.9974691271781921, "reward_std": 6.316715735010803e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009578624740242958, "sampling/sampling_logp_difference/max": 0.9488290548324585, "sampling/importance_sampling_ratio/min": 0.3871941566467285, "sampling/importance_sampling_ratio/mean": 1.001711130142212, "sampling/importance_sampling_ratio/max": 1.6696717739105225, "entropy": 0.04315004916861653, "clip_ratio/low_mean": 0.006177184404805303, "clip_ratio/low_min": 0.006177184404805303, "clip_ratio/high_mean": 0.0036526747280731797, "clip_ratio/high_max": 0.0036526747280731797, "clip_ratio/region_mean": 0.009829859132878482, "reward_total_mean": 0.9974691271781921, "reward_meter_mean": 0.9974691271781921, "reward_meter_std": 6.317263614619151e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9974691271781921, "reward_total_composite_std": 6.317263614619151e-05} {"timestamp_utc": "2026-04-12T00:54:54Z", "mode": "train", "global_step": 1819, "epoch": 0.07306101136683134, "loss": -0.0058, "grad_norm": 2.0940115451812744, "learning_rate": 4.490909090909091e-06, "num_tokens": 4099258.0, "completions/mean_length": 93.75, "completions/min_length": 92.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.75, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.8540617823600769, "rewards/meter/std": 0.2617916762828827, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.8056372404098511, "rewards/total_composite/std": 0.25278210639953613, "reward": 0.8056372404098511, "reward_std": 0.25278210639953613, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.029167180880904198, "sampling/sampling_logp_difference/max": 0.9276943206787109, "sampling/importance_sampling_ratio/min": 0.3954644799232483, "sampling/importance_sampling_ratio/mean": 1.0129790306091309, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23089196532964706, "clip_ratio/low_mean": 0.006778868613764644, "clip_ratio/low_min": 0.006778868613764644, "clip_ratio/high_mean": 0.020909853279590607, "clip_ratio/high_max": 0.020909853279590607, "clip_ratio/region_mean": 0.02768872189335525, "reward_total_mean": 0.8056372404098511, "reward_meter_mean": 0.8540617823600769, "reward_meter_std": 0.2617916762828827, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.8056372404098511, "reward_total_composite_std": 0.25278210639953613} {"timestamp_utc": "2026-04-12T00:54:59Z", "mode": "train", "global_step": 1820, "epoch": 0.0731011768486163, "loss": 0.0047, "grad_norm": 2.0801198482513428, "learning_rate": 4.487878787878788e-06, "num_tokens": 4101001.0, "completions/mean_length": 57.875, "completions/min_length": 57.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.875, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9942807555198669, "rewards/meter/std": 0.0004856908926740289, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9942807555198669, "rewards/total_composite/std": 0.0004856908926740289, "reward": 0.9942807555198669, "reward_std": 0.00048569359933026135, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017049286514520645, "sampling/sampling_logp_difference/max": 1.2708497047424316, "sampling/importance_sampling_ratio/min": 0.2805930972099304, "sampling/importance_sampling_ratio/mean": 1.0053722858428955, "sampling/importance_sampling_ratio/max": 1.5113404989242554, "entropy": 0.10346860811114311, "clip_ratio/low_mean": 0.006392460549250245, "clip_ratio/low_min": 0.006392460549250245, "clip_ratio/high_mean": 0.008658499689772725, "clip_ratio/high_max": 0.008658499689772725, "clip_ratio/region_mean": 0.01505096023902297, "reward_total_mean": 0.9942807555198669, "reward_meter_mean": 0.9942807555198669, "reward_meter_std": 0.0004856908926740289, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9942807555198669, "reward_total_composite_std": 0.0004856908926740289} {"timestamp_utc": "2026-04-12T00:55:03Z", "mode": "train", "global_step": 1821, "epoch": 0.07314134233040126, "loss": 0.0157, "grad_norm": 5.139247417449951, "learning_rate": 4.4848484848484855e-06, "num_tokens": 4102797.0, "completions/mean_length": 74.5, "completions/min_length": 73.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.5, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9978154897689819, "rewards/meter/std": 0.0017045722343027592, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978154897689819, "rewards/total_composite/std": 0.0017045722343027592, "reward": 0.9978154897689819, "reward_std": 0.0017045725835487247, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.055102791637182236, "sampling/sampling_logp_difference/max": 1.0592074394226074, "sampling/importance_sampling_ratio/min": 0.346730500459671, "sampling/importance_sampling_ratio/mean": 1.015410304069519, "sampling/importance_sampling_ratio/max": 1.806069016456604, "entropy": 0.6388323791325092, "clip_ratio/low_mean": 0.016557018272578716, "clip_ratio/low_min": 0.016557018272578716, "clip_ratio/high_mean": 0.040383500047028065, "clip_ratio/high_max": 0.040383500047028065, "clip_ratio/region_mean": 0.05694051831960678, "reward_total_mean": 0.9978154897689819, "reward_meter_mean": 0.9978154897689819, "reward_meter_std": 0.0017045722343027592, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9978154897689819, "reward_total_composite_std": 0.0017045722343027592} {"timestamp_utc": "2026-04-12T00:55:08Z", "mode": "train", "global_step": 1822, "epoch": 0.07318150781218621, "loss": -0.0115, "grad_norm": 5.289615154266357, "learning_rate": 4.481818181818182e-06, "num_tokens": 4104667.0, "completions/mean_length": 73.75, "completions/min_length": 70.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.75, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9970654249191284, "rewards/meter/std": 0.0017285288777202368, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9970654249191284, "rewards/total_composite/std": 0.0017285288777202368, "reward": 0.9970654249191284, "reward_std": 0.0017285366775467992, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05023278295993805, "sampling/sampling_logp_difference/max": 1.2027688026428223, "sampling/importance_sampling_ratio/min": 0.30036142468452454, "sampling/importance_sampling_ratio/mean": 1.0059843063354492, "sampling/importance_sampling_ratio/max": 1.728150725364685, "entropy": 0.5206275209784508, "clip_ratio/low_mean": 0.020897203590720892, "clip_ratio/low_min": 0.020897203590720892, "clip_ratio/high_mean": 0.018180417479015887, "clip_ratio/high_max": 0.018180417479015887, "clip_ratio/region_mean": 0.03907762106973678, "reward_total_mean": 0.9970654249191284, "reward_meter_mean": 0.9970654249191284, "reward_meter_std": 0.0017285288777202368, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9970654249191284, "reward_total_composite_std": 0.0017285288777202368} {"timestamp_utc": "2026-04-12T00:55:12Z", "mode": "train", "global_step": 1823, "epoch": 0.07322167329397117, "loss": -0.0029, "grad_norm": 2.6084089279174805, "learning_rate": 4.478787878787879e-06, "num_tokens": 4106511.0, "completions/mean_length": 64.5, "completions/min_length": 64.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.5, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9984657764434814, "rewards/meter/std": 0.0002891587500926107, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9984657764434814, "rewards/total_composite/std": 0.0002891587500926107, "reward": 0.9984657764434814, "reward_std": 0.0002891614567488432, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007291331887245178, "sampling/sampling_logp_difference/max": 0.45546579360961914, "sampling/importance_sampling_ratio/min": 0.6341525316238403, "sampling/importance_sampling_ratio/mean": 0.9986276626586914, "sampling/importance_sampling_ratio/max": 1.241689920425415, "entropy": 0.033252993831411004, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.007692307699471712, "clip_ratio/high_max": 0.007692307699471712, "clip_ratio/region_mean": 0.007692307699471712, "reward_total_mean": 0.9984657764434814, "reward_meter_mean": 0.9984657764434814, "reward_meter_std": 0.0002891587500926107, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9984657764434814, "reward_total_composite_std": 0.0002891587500926107} {"timestamp_utc": "2026-04-12T00:55:17Z", "mode": "train", "global_step": 1824, "epoch": 0.07326183877575612, "loss": -0.0206, "grad_norm": 11.332904815673828, "learning_rate": 4.4757575757575765e-06, "num_tokens": 4108093.0, "completions/mean_length": 37.75, "completions/min_length": 35.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.75, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.9941079616546631, "rewards/meter/std": 0.012935124337673187, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9941079616546631, "rewards/total_composite/std": 0.012935124337673187, "reward": 0.9941079616546631, "reward_std": 0.012935123406350613, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0540194995701313, "sampling/sampling_logp_difference/max": 1.3527865409851074, "sampling/importance_sampling_ratio/min": 0.25851887464523315, "sampling/importance_sampling_ratio/mean": 1.0034513473510742, "sampling/importance_sampling_ratio/max": 1.575133204460144, "entropy": 0.39845724031329155, "clip_ratio/low_mean": 0.0035714285913854837, "clip_ratio/low_min": 0.0035714285913854837, "clip_ratio/high_mean": 0.044933649245649576, "clip_ratio/high_max": 0.044933649245649576, "clip_ratio/region_mean": 0.04850507783703506, "reward_total_mean": 0.9941079616546631, "reward_meter_mean": 0.9941079616546631, "reward_meter_std": 0.012935124337673187, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9941079616546631, "reward_total_composite_std": 0.012935124337673187} {"timestamp_utc": "2026-04-12T00:55:22Z", "mode": "train", "global_step": 1825, "epoch": 0.07330200425754108, "loss": 0.0106, "grad_norm": 4.047023296356201, "learning_rate": 4.472727272727273e-06, "num_tokens": 4110202.0, "completions/mean_length": 94.625, "completions/min_length": 90.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.625, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.587390661239624, "rewards/meter/std": 0.44364675879478455, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.554187536239624, "rewards/total_composite/std": 0.42821574211120605, "reward": 0.554187536239624, "reward_std": 0.42821574211120605, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03696272522211075, "sampling/sampling_logp_difference/max": 0.9904708862304688, "sampling/importance_sampling_ratio/min": 0.37140175700187683, "sampling/importance_sampling_ratio/mean": 1.0154355764389038, "sampling/importance_sampling_ratio/max": 1.9479297399520874, "entropy": 0.3725603432394564, "clip_ratio/low_mean": 0.02371319057419896, "clip_ratio/low_min": 0.02371319057419896, "clip_ratio/high_mean": 0.0026315790601074696, "clip_ratio/high_max": 0.0026315790601074696, "clip_ratio/region_mean": 0.02634476963430643, "reward_total_mean": 0.554187536239624, "reward_meter_mean": 0.587390661239624, "reward_meter_std": 0.44364675879478455, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.554187536239624, "reward_total_composite_std": 0.42821574211120605} {"timestamp_utc": "2026-04-12T00:55:27Z", "mode": "train", "global_step": 1826, "epoch": 0.07334216973932603, "loss": 0.0078, "grad_norm": 3.477172613143921, "learning_rate": 4.46969696969697e-06, "num_tokens": 4112208.0, "completions/mean_length": 92.75, "completions/min_length": 91.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.75, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.9020949602127075, "rewards/meter/std": 0.12145007401704788, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.8535131216049194, "rewards/total_composite/std": 0.12295859307050705, "reward": 0.8535131216049194, "reward_std": 0.12295857816934586, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02582640014588833, "sampling/sampling_logp_difference/max": 1.0430831909179688, "sampling/importance_sampling_ratio/min": 0.3523665964603424, "sampling/importance_sampling_ratio/mean": 1.0086493492126465, "sampling/importance_sampling_ratio/max": 1.6056605577468872, "entropy": 0.20322295650839806, "clip_ratio/low_mean": 0.0066495382925495505, "clip_ratio/low_min": 0.0066495382925495505, "clip_ratio/high_mean": 0.004032257944345474, "clip_ratio/high_max": 0.004032257944345474, "clip_ratio/region_mean": 0.010681796236895025, "reward_total_mean": 0.8535131216049194, "reward_meter_mean": 0.9020949602127075, "reward_meter_std": 0.12145007401704788, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.8535131216049194, "reward_total_composite_std": 0.12295859307050705} {"timestamp_utc": "2026-04-12T00:55:32Z", "mode": "train", "global_step": 1827, "epoch": 0.07338233522111098, "loss": -0.0104, "grad_norm": 3.58325457572937, "learning_rate": 4.4666666666666665e-06, "num_tokens": 4114241.0, "completions/mean_length": 100.125, "completions/min_length": 96.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.125, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.7311835289001465, "rewards/meter/std": 0.286437064409256, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.7074779868125916, "rewards/total_composite/std": 0.27346280217170715, "reward": 0.7074779868125916, "reward_std": 0.27346280217170715, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04709630832076073, "sampling/sampling_logp_difference/max": 0.8693370819091797, "sampling/importance_sampling_ratio/min": 0.4192293882369995, "sampling/importance_sampling_ratio/mean": 1.018310546875, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37415438517928123, "clip_ratio/low_mean": 0.015318386955186725, "clip_ratio/low_min": 0.015318386955186725, "clip_ratio/high_mean": 0.03203791263513267, "clip_ratio/high_max": 0.03203791263513267, "clip_ratio/region_mean": 0.047356299590319395, "reward_total_mean": 0.7074779868125916, "reward_meter_mean": 0.7311835289001465, "reward_meter_std": 0.286437064409256, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.7074779868125916, "reward_total_composite_std": 0.27346280217170715} {"timestamp_utc": "2026-04-12T00:55:36Z", "mode": "train", "global_step": 1828, "epoch": 0.07342250070289594, "loss": 0.0007, "grad_norm": 1.8939130306243896, "learning_rate": 4.463636363636364e-06, "num_tokens": 4116559.0, "completions/mean_length": 101.75, "completions/min_length": 99.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.75, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.9971822500228882, "rewards/meter/std": 0.0007533096941187978, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9971822500228882, "rewards/total_composite/std": 0.0007533096941187978, "reward": 0.9971822500228882, "reward_std": 0.0007533166790381074, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011065986938774586, "sampling/sampling_logp_difference/max": 0.6848697662353516, "sampling/importance_sampling_ratio/min": 0.5041558742523193, "sampling/importance_sampling_ratio/mean": 1.002759337425232, "sampling/importance_sampling_ratio/max": 1.46595299243927, "entropy": 0.06690388498827815, "clip_ratio/low_mean": 0.0012254902394488454, "clip_ratio/low_min": 0.0012254902394488454, "clip_ratio/high_mean": 0.007329145446419716, "clip_ratio/high_max": 0.007329145446419716, "clip_ratio/region_mean": 0.008554635685868561, "reward_total_mean": 0.9971822500228882, "reward_meter_mean": 0.9971822500228882, "reward_meter_std": 0.0007533096941187978, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9971822500228882, "reward_total_composite_std": 0.0007533096941187978} {"timestamp_utc": "2026-04-12T00:55:42Z", "mode": "train", "global_step": 1829, "epoch": 0.07346266618468089, "loss": -0.0037, "grad_norm": 6.232636451721191, "learning_rate": 4.460606060606061e-06, "num_tokens": 4119597.0, "completions/mean_length": 185.75, "completions/min_length": 183.0, "completions/max_length": 189.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 185.75, "completions/min_terminated_length": 183.0, "completions/max_terminated_length": 189.0, "rewards/meter/mean": 0.9841963052749634, "rewards/meter/std": 0.033278584480285645, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7750000357627869, "rewards/repeat_penalty/std": 0.0707106813788414, "rewards/total_composite/mean": 0.6353453397750854, "rewards/total_composite/std": 0.05924127250909805, "reward": 0.6353453397750854, "reward_std": 0.05924128741025925, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0172612052410841, "sampling/sampling_logp_difference/max": 1.5019760131835938, "sampling/importance_sampling_ratio/min": 0.222689688205719, "sampling/importance_sampling_ratio/mean": 1.0010361671447754, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10350964730605483, "clip_ratio/low_mean": 0.002019958512391895, "clip_ratio/low_min": 0.002019958512391895, "clip_ratio/high_mean": 0.0033174321288242936, "clip_ratio/high_max": 0.0033174321288242936, "clip_ratio/region_mean": 0.005337390641216189, "reward_total_mean": 0.6353453397750854, "reward_meter_mean": 0.9841963052749634, "reward_meter_std": 0.033278584480285645, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7750000357627869, "reward_repeat_penalty_std": 0.0707106813788414, "reward_total_composite_mean": 0.6353453397750854, "reward_total_composite_std": 0.05924127250909805} {"timestamp_utc": "2026-04-12T00:55:47Z", "mode": "train", "global_step": 1830, "epoch": 0.07350283166646585, "loss": 0.0001, "grad_norm": 0.13791632652282715, "learning_rate": 4.4575757575757575e-06, "num_tokens": 4121701.0, "completions/mean_length": 99.0, "completions/min_length": 99.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.0, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.9990502595901489, "rewards/meter/std": 1.3007632333028596e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990502595901489, "rewards/total_composite/std": 1.3007632333028596e-05, "reward": 0.9990502595901489, "reward_std": 1.300759322475642e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00393805792555213, "sampling/sampling_logp_difference/max": 0.512671709060669, "sampling/importance_sampling_ratio/min": 0.6270684599876404, "sampling/importance_sampling_ratio/mean": 1.0018277168273926, "sampling/importance_sampling_ratio/max": 1.6697462797164917, "entropy": 0.020989181706681848, "clip_ratio/low_mean": 0.0012626262614503503, "clip_ratio/low_min": 0.0012626262614503503, "clip_ratio/high_mean": 0.0012626262614503503, "clip_ratio/high_max": 0.0012626262614503503, "clip_ratio/region_mean": 0.0025252525229007006, "reward_total_mean": 0.9990502595901489, "reward_meter_mean": 0.9990502595901489, "reward_meter_std": 1.3007632333028596e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990502595901489, "reward_total_composite_std": 1.3007632333028596e-05} {"timestamp_utc": "2026-04-12T00:55:53Z", "mode": "train", "global_step": 1831, "epoch": 0.0735429971482508, "loss": 0.0079, "grad_norm": 3.588621139526367, "learning_rate": 4.454545454545455e-06, "num_tokens": 4123721.0, "completions/mean_length": 89.5, "completions/min_length": 85.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.5, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9949823021888733, "rewards/meter/std": 0.0008774827001616359, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9949823021888733, "rewards/total_composite/std": 0.0008774827001616359, "reward": 0.9949823021888733, "reward_std": 0.0008774733869358897, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02186514623463154, "sampling/sampling_logp_difference/max": 1.0048408508300781, "sampling/importance_sampling_ratio/min": 0.3661029040813446, "sampling/importance_sampling_ratio/mean": 1.007697343826294, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.14992150850594044, "clip_ratio/low_mean": 0.010989011265337467, "clip_ratio/low_min": 0.010989011265337467, "clip_ratio/high_mean": 0.0055555556900799274, "clip_ratio/high_max": 0.0055555556900799274, "clip_ratio/region_mean": 0.016544566955417395, "reward_total_mean": 0.9949823021888733, "reward_meter_mean": 0.9949823021888733, "reward_meter_std": 0.0008774827001616359, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9949823021888733, "reward_total_composite_std": 0.0008774827001616359} {"timestamp_utc": "2026-04-12T00:55:58Z", "mode": "train", "global_step": 1832, "epoch": 0.07358316263003575, "loss": 0.0014, "grad_norm": 1.7238988876342773, "learning_rate": 4.451515151515152e-06, "num_tokens": 4125807.0, "completions/mean_length": 89.75, "completions/min_length": 88.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.75, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9952454566955566, "rewards/meter/std": 0.00010376512364018708, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9952454566955566, "rewards/total_composite/std": 0.00010376512364018708, "reward": 0.9952454566955566, "reward_std": 0.00010377082071499899, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014209388755261898, "sampling/sampling_logp_difference/max": 1.3864455223083496, "sampling/importance_sampling_ratio/min": 0.2499622255563736, "sampling/importance_sampling_ratio/mean": 1.001192331314087, "sampling/importance_sampling_ratio/max": 1.3277589082717896, "entropy": 0.09856001799926162, "clip_ratio/low_mean": 0.008428030530922115, "clip_ratio/low_min": 0.008428030530922115, "clip_ratio/high_mean": 0.0055555556900799274, "clip_ratio/high_max": 0.0055555556900799274, "clip_ratio/region_mean": 0.013983586221002042, "reward_total_mean": 0.9952454566955566, "reward_meter_mean": 0.9952454566955566, "reward_meter_std": 0.00010376512364018708, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9952454566955566, "reward_total_composite_std": 0.00010376512364018708} {"timestamp_utc": "2026-04-12T00:56:02Z", "mode": "train", "global_step": 1833, "epoch": 0.07362332811182071, "loss": 0.0092, "grad_norm": 3.797224283218384, "learning_rate": 4.448484848484848e-06, "num_tokens": 4127747.0, "completions/mean_length": 75.5, "completions/min_length": 72.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.5, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9976780414581299, "rewards/meter/std": 0.0009906893828883767, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9976780414581299, "rewards/total_composite/std": 0.0009906893828883767, "reward": 0.9976780414581299, "reward_std": 0.0009906899649649858, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03965409845113754, "sampling/sampling_logp_difference/max": 0.8967013359069824, "sampling/importance_sampling_ratio/min": 0.4079129993915558, "sampling/importance_sampling_ratio/mean": 1.0070687532424927, "sampling/importance_sampling_ratio/max": 1.881462812423706, "entropy": 0.36707402765750885, "clip_ratio/low_mean": 0.018417607876472175, "clip_ratio/low_min": 0.018417607876472175, "clip_ratio/high_mean": 0.016408111667260528, "clip_ratio/high_max": 0.016408111667260528, "clip_ratio/region_mean": 0.0348257195437327, "reward_total_mean": 0.9976780414581299, "reward_meter_mean": 0.9976780414581299, "reward_meter_std": 0.0009906893828883767, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9976780414581299, "reward_total_composite_std": 0.0009906893828883767} {"timestamp_utc": "2026-04-12T00:56:08Z", "mode": "train", "global_step": 1834, "epoch": 0.07366349359360566, "loss": -0.0051, "grad_norm": 3.744053840637207, "learning_rate": 4.445454545454546e-06, "num_tokens": 4130264.0, "completions/mean_length": 133.625, "completions/min_length": 130.0, "completions/max_length": 137.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 133.625, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.9959684014320374, "rewards/meter/std": 0.0031451100949198008, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9285714626312256, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.9247440099716187, "rewards/total_composite/std": 0.07493867725133896, "reward": 0.9247440099716187, "reward_std": 0.07493869215250015, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06190461292862892, "sampling/sampling_logp_difference/max": 2.0104918479919434, "sampling/importance_sampling_ratio/min": 0.1339227855205536, "sampling/importance_sampling_ratio/mean": 1.0151046514511108, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.501116044819355, "clip_ratio/low_mean": 0.015092328016180545, "clip_ratio/low_min": 0.015092328016180545, "clip_ratio/high_mean": 0.03248523222282529, "clip_ratio/high_max": 0.03248523222282529, "clip_ratio/region_mean": 0.047577560239005834, "reward_total_mean": 0.9247440099716187, "reward_meter_mean": 0.9959684014320374, "reward_meter_std": 0.0031451100949198008, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9285714626312256, "reward_repeat_penalty_std": 0.07636035233736038, "reward_total_composite_mean": 0.9247440099716187, "reward_total_composite_std": 0.07493867725133896} {"timestamp_utc": "2026-04-12T00:56:12Z", "mode": "train", "global_step": 1835, "epoch": 0.07370365907539062, "loss": -0.0192, "grad_norm": 5.053841590881348, "learning_rate": 4.442424242424243e-06, "num_tokens": 4132027.0, "completions/mean_length": 62.375, "completions/min_length": 58.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.375, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.8166337013244629, "rewards/meter/std": 0.3090912401676178, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8166337013244629, "rewards/total_composite/std": 0.3090912401676178, "reward": 0.8166337013244629, "reward_std": 0.3090912103652954, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.038962990045547485, "sampling/sampling_logp_difference/max": 1.3373403549194336, "sampling/importance_sampling_ratio/min": 0.2625430226325989, "sampling/importance_sampling_ratio/mean": 0.9984228610992432, "sampling/importance_sampling_ratio/max": 1.7029978036880493, "entropy": 0.26288264244794846, "clip_ratio/low_mean": 0.008216594811528921, "clip_ratio/low_min": 0.008216594811528921, "clip_ratio/high_mean": 0.023937532445415854, "clip_ratio/high_max": 0.023937532445415854, "clip_ratio/region_mean": 0.032154127256944776, "reward_total_mean": 0.8166337013244629, "reward_meter_mean": 0.8166337013244629, "reward_meter_std": 0.3090912401676178, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8166337013244629, "reward_total_composite_std": 0.3090912401676178} {"timestamp_utc": "2026-04-12T00:56:17Z", "mode": "train", "global_step": 1836, "epoch": 0.07374382455717557, "loss": 0.0132, "grad_norm": 5.317627906799316, "learning_rate": 4.43939393939394e-06, "num_tokens": 4134279.0, "completions/mean_length": 116.5, "completions/min_length": 111.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.5, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.9806186556816101, "rewards/meter/std": 0.017093582078814507, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9559872150421143, "rewards/total_composite/std": 0.06989036500453949, "reward": 0.9559872150421143, "reward_std": 0.06989037245512009, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07094138115644455, "sampling/sampling_logp_difference/max": 1.1895244121551514, "sampling/importance_sampling_ratio/min": 0.30436596274375916, "sampling/importance_sampling_ratio/mean": 1.0162169933319092, "sampling/importance_sampling_ratio/max": 1.9605425596237183, "entropy": 0.6946443617343903, "clip_ratio/low_mean": 0.00966143561527133, "clip_ratio/low_min": 0.00966143561527133, "clip_ratio/high_mean": 0.04100045142695308, "clip_ratio/high_max": 0.04100045142695308, "clip_ratio/region_mean": 0.05066188704222441, "reward_total_mean": 0.9559872150421143, "reward_meter_mean": 0.9806186556816101, "reward_meter_std": 0.017093582078814507, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9559872150421143, "reward_total_composite_std": 0.06989036500453949} {"timestamp_utc": "2026-04-12T00:56:23Z", "mode": "train", "global_step": 1837, "epoch": 0.07378399003896052, "loss": 0.0003, "grad_norm": 1.2245804071426392, "learning_rate": 4.436363636363637e-06, "num_tokens": 4137119.0, "completions/mean_length": 170.0, "completions/min_length": 170.0, "completions/max_length": 170.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 170.0, "completions/min_terminated_length": 170.0, "completions/max_terminated_length": 170.0, "rewards/meter/mean": 0.9958264231681824, "rewards/meter/std": 6.799297989346087e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.762499988079071, "rewards/repeat_penalty/std": 0.05175492912530899, "rewards/total_composite/mean": 0.7593173980712891, "rewards/total_composite/std": 0.05153491348028183, "reward": 0.7593173980712891, "reward_std": 0.051534902304410934, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005329389125108719, "sampling/sampling_logp_difference/max": 0.6398299336433411, "sampling/importance_sampling_ratio/min": 0.5273821353912354, "sampling/importance_sampling_ratio/mean": 1.0017375946044922, "sampling/importance_sampling_ratio/max": 1.4358134269714355, "entropy": 0.04040496004745364, "clip_ratio/low_mean": 0.000735294132027775, "clip_ratio/low_min": 0.000735294132027775, "clip_ratio/high_mean": 0.0029411765281111, "clip_ratio/high_max": 0.0029411765281111, "clip_ratio/region_mean": 0.0036764706601388752, "reward_total_mean": 0.7593173980712891, "reward_meter_mean": 0.9958264231681824, "reward_meter_std": 6.799297989346087e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.762499988079071, "reward_repeat_penalty_std": 0.05175492912530899, "reward_total_composite_mean": 0.7593173980712891, "reward_total_composite_std": 0.05153491348028183} {"timestamp_utc": "2026-04-12T00:56:29Z", "mode": "train", "global_step": 1838, "epoch": 0.07382415552074548, "loss": -0.019, "grad_norm": 3.743595838546753, "learning_rate": 4.433333333333334e-06, "num_tokens": 4140579.0, "completions/mean_length": 228.5, "completions/min_length": 195.0, "completions/max_length": 240.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 228.5, "completions/min_terminated_length": 195.0, "completions/max_terminated_length": 240.0, "rewards/meter/mean": 0.7707350254058838, "rewards/meter/std": 0.3024085462093353, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9791666865348816, "rewards/repeat_penalty/std": 0.038575831800699234, "rewards/total_composite/mean": 0.7390586137771606, "rewards/total_composite/std": 0.29844897985458374, "reward": 0.7390586137771606, "reward_std": 0.29844897985458374, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06499035656452179, "sampling/sampling_logp_difference/max": 1.2022161483764648, "sampling/importance_sampling_ratio/min": 0.3005274534225464, "sampling/importance_sampling_ratio/mean": 1.015356183052063, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.668959241360426, "clip_ratio/low_mean": 0.022611827589571476, "clip_ratio/low_min": 0.022611827589571476, "clip_ratio/high_mean": 0.022940170136280358, "clip_ratio/high_max": 0.022940170136280358, "clip_ratio/region_mean": 0.045551997725851834, "reward_total_mean": 0.7390586137771606, "reward_meter_mean": 0.7707350254058838, "reward_meter_std": 0.3024085462093353, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9791666865348816, "reward_repeat_penalty_std": 0.038575831800699234, "reward_total_composite_mean": 0.7390586137771606, "reward_total_composite_std": 0.29844897985458374} {"timestamp_utc": "2026-04-12T00:56:34Z", "mode": "train", "global_step": 1839, "epoch": 0.07386432100253043, "loss": -0.03, "grad_norm": 5.02839994430542, "learning_rate": 4.430303030303031e-06, "num_tokens": 4142461.0, "completions/mean_length": 78.25, "completions/min_length": 73.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.25, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.9820351600646973, "rewards/meter/std": 0.018020369112491608, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9820351600646973, "rewards/total_composite/std": 0.018020369112491608, "reward": 0.9820351600646973, "reward_std": 0.018020374700427055, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07023806869983673, "sampling/sampling_logp_difference/max": 1.4098834991455078, "sampling/importance_sampling_ratio/min": 0.24417172372341156, "sampling/importance_sampling_ratio/mean": 1.0230660438537598, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7259627729654312, "clip_ratio/low_mean": 0.01651052851229906, "clip_ratio/low_min": 0.01651052851229906, "clip_ratio/high_mean": 0.02181874285452068, "clip_ratio/high_max": 0.02181874285452068, "clip_ratio/region_mean": 0.03832927136681974, "reward_total_mean": 0.9820351600646973, "reward_meter_mean": 0.9820351600646973, "reward_meter_std": 0.018020369112491608, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9820351600646973, "reward_total_composite_std": 0.018020369112491608} {"timestamp_utc": "2026-04-12T00:56:39Z", "mode": "train", "global_step": 1840, "epoch": 0.07390448648431538, "loss": 0.0184, "grad_norm": 6.407735347747803, "learning_rate": 4.4272727272727275e-06, "num_tokens": 4144162.0, "completions/mean_length": 49.625, "completions/min_length": 46.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.625, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.7431851029396057, "rewards/meter/std": 0.294132262468338, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7431851029396057, "rewards/total_composite/std": 0.294132262468338, "reward": 0.7431851029396057, "reward_std": 0.2941322326660156, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05073566362261772, "sampling/sampling_logp_difference/max": 1.3865375518798828, "sampling/importance_sampling_ratio/min": 0.2499392181634903, "sampling/importance_sampling_ratio/mean": 1.0074049234390259, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.24762486666440964, "clip_ratio/low_mean": 0.022959183901548386, "clip_ratio/low_min": 0.022959183901548386, "clip_ratio/high_mean": 0.027999775717034936, "clip_ratio/high_max": 0.027999775717034936, "clip_ratio/region_mean": 0.05095895961858332, "reward_total_mean": 0.7431851029396057, "reward_meter_mean": 0.7431851029396057, "reward_meter_std": 0.294132262468338, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7431851029396057, "reward_total_composite_std": 0.294132262468338} {"timestamp_utc": "2026-04-12T00:56:44Z", "mode": "train", "global_step": 1841, "epoch": 0.07394465196610034, "loss": 0.012, "grad_norm": 3.6605427265167236, "learning_rate": 4.424242424242425e-06, "num_tokens": 4146697.0, "completions/mean_length": 119.875, "completions/min_length": 116.0, "completions/max_length": 123.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 119.875, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.9946545362472534, "rewards/meter/std": 0.0017242503818124533, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.910714328289032, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9058225154876709, "rewards/total_composite/std": 0.07321521639823914, "reward": 0.9058225154876709, "reward_std": 0.07321520894765854, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016406936571002007, "sampling/sampling_logp_difference/max": 1.0592851638793945, "sampling/importance_sampling_ratio/min": 0.34670352935791016, "sampling/importance_sampling_ratio/mean": 0.9984482526779175, "sampling/importance_sampling_ratio/max": 1.6830854415893555, "entropy": 0.08843067381531, "clip_ratio/low_mean": 0.008309659664519131, "clip_ratio/low_min": 0.008309659664519131, "clip_ratio/high_mean": 0.00737299001775682, "clip_ratio/high_max": 0.00737299001775682, "clip_ratio/region_mean": 0.01568264968227595, "reward_total_mean": 0.9058225154876709, "reward_meter_mean": 0.9946545362472534, "reward_meter_std": 0.0017242503818124533, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.910714328289032, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.9058225154876709, "reward_total_composite_std": 0.07321521639823914} {"timestamp_utc": "2026-04-12T00:56:48Z", "mode": "train", "global_step": 1842, "epoch": 0.07398481744788529, "loss": 0.0062, "grad_norm": 4.1317853927612305, "learning_rate": 4.421212121212122e-06, "num_tokens": 4148545.0, "completions/mean_length": 65.0, "completions/min_length": 63.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.0, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.8449044227600098, "rewards/meter/std": 0.19496634602546692, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8449044227600098, "rewards/total_composite/std": 0.19496634602546692, "reward": 0.8449044227600098, "reward_std": 0.19496633112430573, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03727368265390396, "sampling/sampling_logp_difference/max": 1.3517389297485352, "sampling/importance_sampling_ratio/min": 0.2587898373603821, "sampling/importance_sampling_ratio/mean": 1.0063512325286865, "sampling/importance_sampling_ratio/max": 1.8676364421844482, "entropy": 0.2709883898496628, "clip_ratio/low_mean": 0.012987380847334862, "clip_ratio/low_min": 0.012987380847334862, "clip_ratio/high_mean": 0.00787545822095126, "clip_ratio/high_max": 0.00787545822095126, "clip_ratio/region_mean": 0.02086283906828612, "reward_total_mean": 0.8449044227600098, "reward_meter_mean": 0.8449044227600098, "reward_meter_std": 0.19496634602546692, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8449044227600098, "reward_total_composite_std": 0.19496634602546692} {"timestamp_utc": "2026-04-12T00:56:53Z", "mode": "train", "global_step": 1843, "epoch": 0.07402498292967025, "loss": 0.0033, "grad_norm": 2.7024903297424316, "learning_rate": 4.418181818181818e-06, "num_tokens": 4150316.0, "completions/mean_length": 69.375, "completions/min_length": 69.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.375, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9967325925827026, "rewards/meter/std": 0.0019061544444411993, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9967325925827026, "rewards/total_composite/std": 0.0019061544444411993, "reward": 0.9967325925827026, "reward_std": 0.0019061638740822673, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015712035819888115, "sampling/sampling_logp_difference/max": 1.4424948692321777, "sampling/importance_sampling_ratio/min": 0.23633739352226257, "sampling/importance_sampling_ratio/mean": 1.0016083717346191, "sampling/importance_sampling_ratio/max": 1.3403892517089844, "entropy": 0.09468106366693974, "clip_ratio/low_mean": 0.005357142887078226, "clip_ratio/low_min": 0.005357142887078226, "clip_ratio/high_mean": 0.005434782709926367, "clip_ratio/high_max": 0.005434782709926367, "clip_ratio/region_mean": 0.010791925597004592, "reward_total_mean": 0.9967325925827026, "reward_meter_mean": 0.9967325925827026, "reward_meter_std": 0.0019061544444411993, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9967325925827026, "reward_total_composite_std": 0.0019061544444411993} {"timestamp_utc": "2026-04-12T00:56:58Z", "mode": "train", "global_step": 1844, "epoch": 0.0740651484114552, "loss": 0.0055, "grad_norm": 4.094827175140381, "learning_rate": 4.415151515151516e-06, "num_tokens": 4152604.0, "completions/mean_length": 94.0, "completions/min_length": 91.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.0, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.9527285695075989, "rewards/meter/std": 0.020587077364325523, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9527285695075989, "rewards/total_composite/std": 0.020587077364325523, "reward": 0.9527285695075989, "reward_std": 0.02058705873787403, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0326327309012413, "sampling/sampling_logp_difference/max": 2.7240052223205566, "sampling/importance_sampling_ratio/min": 0.06561143696308136, "sampling/importance_sampling_ratio/mean": 1.006330132484436, "sampling/importance_sampling_ratio/max": 1.6599549055099487, "entropy": 0.2582856100052595, "clip_ratio/low_mean": 0.00394825276453048, "clip_ratio/low_min": 0.00394825276453048, "clip_ratio/high_mean": 0.01599274785257876, "clip_ratio/high_max": 0.01599274785257876, "clip_ratio/region_mean": 0.01994100061710924, "reward_total_mean": 0.9527285695075989, "reward_meter_mean": 0.9527285695075989, "reward_meter_std": 0.020587077364325523, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9527285695075989, "reward_total_composite_std": 0.020587077364325523} {"timestamp_utc": "2026-04-12T00:57:02Z", "mode": "train", "global_step": 1845, "epoch": 0.07410531389324015, "loss": -0.0044, "grad_norm": 6.061025619506836, "learning_rate": 4.412121212121213e-06, "num_tokens": 4154452.0, "completions/mean_length": 69.0, "completions/min_length": 68.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.0, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9951037168502808, "rewards/meter/std": 0.0059710158966481686, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9951037168502808, "rewards/total_composite/std": 0.0059710158966481686, "reward": 0.9951037168502808, "reward_std": 0.005971014499664307, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009792874567210674, "sampling/sampling_logp_difference/max": 0.5869016647338867, "sampling/importance_sampling_ratio/min": 0.5560474395751953, "sampling/importance_sampling_ratio/mean": 1.0022023916244507, "sampling/importance_sampling_ratio/max": 1.1276072263717651, "entropy": 0.07721387594938278, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.005434782709926367, "clip_ratio/high_max": 0.005434782709926367, "clip_ratio/region_mean": 0.005434782709926367, "reward_total_mean": 0.9951037168502808, "reward_meter_mean": 0.9951037168502808, "reward_meter_std": 0.0059710158966481686, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9951037168502808, "reward_total_composite_std": 0.0059710158966481686} {"timestamp_utc": "2026-04-12T00:57:08Z", "mode": "train", "global_step": 1846, "epoch": 0.07414547937502511, "loss": 0.0137, "grad_norm": 2.195159912109375, "learning_rate": 4.409090909090909e-06, "num_tokens": 4157062.0, "completions/mean_length": 152.25, "completions/min_length": 150.0, "completions/max_length": 156.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 152.25, "completions/min_terminated_length": 150.0, "completions/max_terminated_length": 156.0, "rewards/meter/mean": 0.8805509805679321, "rewards/meter/std": 0.11502306908369064, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8611111640930176, "rewards/repeat_penalty/std": 0.05143444612622261, "rewards/total_composite/mean": 0.7611056566238403, "rewards/total_composite/std": 0.12567628920078278, "reward": 0.7611056566238403, "reward_std": 0.12567628920078278, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02993069775402546, "sampling/sampling_logp_difference/max": 1.2740850448608398, "sampling/importance_sampling_ratio/min": 0.27968674898147583, "sampling/importance_sampling_ratio/mean": 1.0067634582519531, "sampling/importance_sampling_ratio/max": 1.9269274473190308, "entropy": 0.28932092152535915, "clip_ratio/low_mean": 0.00973088713362813, "clip_ratio/low_min": 0.00973088713362813, "clip_ratio/high_mean": 0.015674303169362247, "clip_ratio/high_max": 0.015674303169362247, "clip_ratio/region_mean": 0.025405190302990377, "reward_total_mean": 0.7611056566238403, "reward_meter_mean": 0.8805509805679321, "reward_meter_std": 0.11502306908369064, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8611111640930176, "reward_repeat_penalty_std": 0.05143444612622261, "reward_total_composite_mean": 0.7611056566238403, "reward_total_composite_std": 0.12567628920078278} {"timestamp_utc": "2026-04-12T00:57:13Z", "mode": "train", "global_step": 1847, "epoch": 0.07418564485681006, "loss": 0.002, "grad_norm": 1.8122365474700928, "learning_rate": 4.4060606060606066e-06, "num_tokens": 4159504.0, "completions/mean_length": 137.25, "completions/min_length": 137.0, "completions/max_length": 138.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 137.25, "completions/min_terminated_length": 137.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.9956631660461426, "rewards/meter/std": 0.004949868656694889, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.8709865212440491, "rewards/total_composite/std": 0.04542805626988411, "reward": 0.8709865212440491, "reward_std": 0.04542805254459381, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00922767911106348, "sampling/sampling_logp_difference/max": 1.0127711296081543, "sampling/importance_sampling_ratio/min": 0.36321109533309937, "sampling/importance_sampling_ratio/mean": 1.0003859996795654, "sampling/importance_sampling_ratio/max": 1.290838360786438, "entropy": 0.058579874224960804, "clip_ratio/low_mean": 0.0027372262557037175, "clip_ratio/low_min": 0.0027372262557037175, "clip_ratio/high_mean": 0.0018115942366421223, "clip_ratio/high_max": 0.0018115942366421223, "clip_ratio/region_mean": 0.00454882049234584, "reward_total_mean": 0.8709865212440491, "reward_meter_mean": 0.9956631660461426, "reward_meter_std": 0.004949868656694889, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.8709865212440491, "reward_total_composite_std": 0.04542805626988411} {"timestamp_utc": "2026-04-12T00:57:20Z", "mode": "train", "global_step": 1848, "epoch": 0.07422581033859502, "loss": -0.0158, "grad_norm": 3.0020229816436768, "learning_rate": 4.403030303030304e-06, "num_tokens": 4162692.0, "completions/mean_length": 193.5, "completions/min_length": 181.0, "completions/max_length": 226.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 193.5, "completions/min_terminated_length": 181.0, "completions/max_terminated_length": 226.0, "rewards/meter/mean": 0.9245690107345581, "rewards/meter/std": 0.12092939764261246, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9245690107345581, "rewards/total_composite/std": 0.12092939764261246, "reward": 0.9245690107345581, "reward_std": 0.12092941999435425, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06919539719820023, "sampling/sampling_logp_difference/max": 1.4243488311767578, "sampling/importance_sampling_ratio/min": 0.24066513776779175, "sampling/importance_sampling_ratio/mean": 1.01200270652771, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6156773902475834, "clip_ratio/low_mean": 0.01064768130891025, "clip_ratio/low_min": 0.01064768130891025, "clip_ratio/high_mean": 0.03866560058668256, "clip_ratio/high_max": 0.03866560058668256, "clip_ratio/region_mean": 0.04931328189559281, "reward_total_mean": 0.9245690107345581, "reward_meter_mean": 0.9245690107345581, "reward_meter_std": 0.12092939764261246, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9245690107345581, "reward_total_composite_std": 0.12092939764261246} {"timestamp_utc": "2026-04-12T00:57:25Z", "mode": "train", "global_step": 1849, "epoch": 0.07426597582037997, "loss": 0.0019, "grad_norm": 0.7777861952781677, "learning_rate": 4.4e-06, "num_tokens": 4165004.0, "completions/mean_length": 103.0, "completions/min_length": 103.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.0, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.9962564706802368, "rewards/meter/std": 0.003385263029485941, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9962564706802368, "rewards/total_composite/std": 0.003385263029485941, "reward": 0.9962564706802368, "reward_std": 0.0033852593041956425, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006807548459619284, "sampling/sampling_logp_difference/max": 0.6204808950424194, "sampling/importance_sampling_ratio/min": 0.5530079007148743, "sampling/importance_sampling_ratio/mean": 1.0040092468261719, "sampling/importance_sampling_ratio/max": 1.859822154045105, "entropy": 0.04185265023261309, "clip_ratio/low_mean": 0.0012135922443121672, "clip_ratio/low_min": 0.0012135922443121672, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0012135922443121672, "reward_total_mean": 0.9962564706802368, "reward_meter_mean": 0.9962564706802368, "reward_meter_std": 0.003385263029485941, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9962564706802368, "reward_total_composite_std": 0.003385263029485941} {"timestamp_utc": "2026-04-12T00:57:29Z", "mode": "train", "global_step": 1850, "epoch": 0.07430614130216492, "loss": -0.0062, "grad_norm": 5.336696624755859, "learning_rate": 4.3969696969696975e-06, "num_tokens": 4166714.0, "completions/mean_length": 65.75, "completions/min_length": 64.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.75, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9984281063079834, "rewards/meter/std": 0.0004244510782882571, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9984281063079834, "rewards/total_composite/std": 0.0004244510782882571, "reward": 0.9984281063079834, "reward_std": 0.0004244422307237983, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02056110091507435, "sampling/sampling_logp_difference/max": 1.2651333808898926, "sampling/importance_sampling_ratio/min": 0.2822016477584839, "sampling/importance_sampling_ratio/mean": 0.9998464584350586, "sampling/importance_sampling_ratio/max": 1.7678344249725342, "entropy": 0.06614176463335752, "clip_ratio/low_mean": 0.005741003900766373, "clip_ratio/low_min": 0.005741003900766373, "clip_ratio/high_mean": 0.011363636702299118, "clip_ratio/high_max": 0.011363636702299118, "clip_ratio/region_mean": 0.01710464060306549, "reward_total_mean": 0.9984281063079834, "reward_meter_mean": 0.9984281063079834, "reward_meter_std": 0.0004244510782882571, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9984281063079834, "reward_total_composite_std": 0.0004244510782882571} {"timestamp_utc": "2026-04-12T00:58:32Z", "mode": "eval", "global_step": 1850, "epoch": 0.07430614130216492, "eval_loss": NaN, "eval_runtime": 62.689, "eval_samples_per_second": 1.659, "eval_steps_per_second": 0.207, "eval_num_tokens": 4166714.0, "eval_completions/mean_length": 193.1153846153846, "eval_completions/min_length": 63.76923076923077, "eval_completions/max_length": 327.2307692307692, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 193.1153846153846, "eval_completions/min_terminated_length": 63.76923076923077, "eval_completions/max_terminated_length": 327.2307692307692, "eval_rewards/meter/mean": 0.6479656077348269, "eval_rewards/meter/std": 0.4532018624819242, "eval_rewards/count_adherence/mean": 0.8968340616959792, "eval_rewards/count_adherence/std": 0.13571923226118088, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/repeat_penalty/mean": 0.8583002182153555, "eval_rewards/repeat_penalty/std": 0.13356439138834292, "eval_rewards/total_composite/mean": 0.510624709037634, "eval_rewards/total_composite/std": 0.38263802230358124, "eval_reward": 0.510624709037634, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.023962400352152493, "eval_sampling/sampling_logp_difference/max": 1.0324598183998694, "eval_sampling/importance_sampling_ratio/min": 0.3671792642428325, "eval_sampling/importance_sampling_ratio/mean": 1.0072369300402129, "eval_sampling/importance_sampling_ratio/max": 1.4861979392858653, "eval_entropy": 0.2572394087910652, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.510624709037634, "eval_reward_meter_mean": 0.6479656077348269, "eval_reward_meter_std": 0.4532018624819242, "eval_reward_count_adherence_mean": 0.8968340616959792, "eval_reward_count_adherence_std": 0.13571923226118088, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_repeat_penalty_mean": 0.8583002182153555, "eval_reward_repeat_penalty_std": 0.13356439138834292, "eval_reward_total_composite_mean": 0.510624709037634, "eval_reward_total_composite_std": 0.38263802230358124} {"timestamp_utc": "2026-04-12T00:58:40Z", "mode": "train", "global_step": 1851, "epoch": 0.07434630678394988, "loss": 0.0113, "grad_norm": 4.0140767097473145, "learning_rate": 4.393939393939394e-06, "num_tokens": 4168913.0, "completions/mean_length": 110.875, "completions/min_length": 103.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 110.875, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.37161165475845337, "rewards/meter/std": 0.27739158272743225, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.37161165475845337, "rewards/total_composite/std": 0.27739158272743225, "reward": 0.37161165475845337, "reward_std": 0.27739158272743225, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06999441236257553, "sampling/sampling_logp_difference/max": 1.3712263107299805, "sampling/importance_sampling_ratio/min": 0.2537955343723297, "sampling/importance_sampling_ratio/mean": 1.0157358646392822, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5153284594416618, "clip_ratio/low_mean": 0.044483832316473126, "clip_ratio/low_min": 0.044483832316473126, "clip_ratio/high_mean": 0.00905440840870142, "clip_ratio/high_max": 0.00905440840870142, "clip_ratio/region_mean": 0.053538240725174546, "reward_total_mean": 0.37161165475845337, "reward_meter_mean": 0.37161165475845337, "reward_meter_std": 0.27739158272743225, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.37161165475845337, "reward_total_composite_std": 0.27739158272743225} {"timestamp_utc": "2026-04-12T00:58:47Z", "mode": "train", "global_step": 1852, "epoch": 0.07438647226573483, "loss": 0.0023, "grad_norm": 3.773585557937622, "learning_rate": 4.390909090909091e-06, "num_tokens": 4172359.0, "completions/mean_length": 247.75, "completions/min_length": 222.0, "completions/max_length": 273.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 247.75, "completions/min_terminated_length": 222.0, "completions/max_terminated_length": 273.0, "rewards/meter/mean": 0.6540890336036682, "rewards/meter/std": 0.43809670209884644, "rewards/count_adherence/mean": 0.9464285969734192, "rewards/count_adherence/std": 0.07393559068441391, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9910714626312256, "rewards/repeat_penalty/std": 0.025253823027014732, "rewards/total_composite/mean": 0.6160822510719299, "rewards/total_composite/std": 0.42173710465431213, "reward": 0.6160822510719299, "reward_std": 0.42173710465431213, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09465256333351135, "sampling/sampling_logp_difference/max": 5.8384318351745605, "sampling/importance_sampling_ratio/min": 0.002913407515734434, "sampling/importance_sampling_ratio/mean": 1.012731671333313, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7441882863640785, "clip_ratio/low_mean": 0.01895719300955534, "clip_ratio/low_min": 0.01895719300955534, "clip_ratio/high_mean": 0.04731742758303881, "clip_ratio/high_max": 0.04731742758303881, "clip_ratio/region_mean": 0.06627462059259415, "reward_total_mean": 0.6160822510719299, "reward_meter_mean": 0.6540890336036682, "reward_meter_std": 0.43809670209884644, "reward_count_adherence_mean": 0.9464285969734192, "reward_count_adherence_std": 0.07393559068441391, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9910714626312256, "reward_repeat_penalty_std": 0.025253823027014732, "reward_total_composite_mean": 0.6160822510719299, "reward_total_composite_std": 0.42173710465431213} {"timestamp_utc": "2026-04-12T00:58:51Z", "mode": "train", "global_step": 1853, "epoch": 0.07442663774751979, "loss": 0.006, "grad_norm": 11.522212982177734, "learning_rate": 4.387878787878788e-06, "num_tokens": 4174010.0, "completions/mean_length": 35.375, "completions/min_length": 35.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.375, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9776397943496704, "rewards/meter/std": 0.01006159745156765, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9776397943496704, "rewards/total_composite/std": 0.01006159745156765, "reward": 0.9776397943496704, "reward_std": 0.010061599314212799, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01866281032562256, "sampling/sampling_logp_difference/max": 0.5338115692138672, "sampling/importance_sampling_ratio/min": 0.6540494561195374, "sampling/importance_sampling_ratio/mean": 1.0117498636245728, "sampling/importance_sampling_ratio/max": 1.7054202556610107, "entropy": 0.18969010561704636, "clip_ratio/low_mean": 0.0035714285913854837, "clip_ratio/low_min": 0.0035714285913854837, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0035714285913854837, "reward_total_mean": 0.9776397943496704, "reward_meter_mean": 0.9776397943496704, "reward_meter_std": 0.01006159745156765, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9776397943496704, "reward_total_composite_std": 0.01006159745156765} {"timestamp_utc": "2026-04-12T00:58:57Z", "mode": "train", "global_step": 1854, "epoch": 0.07446680322930474, "loss": 0.0082, "grad_norm": 4.39829158782959, "learning_rate": 4.384848484848485e-06, "num_tokens": 4176279.0, "completions/mean_length": 114.625, "completions/min_length": 110.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.625, "completions/min_terminated_length": 110.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.9875234365463257, "rewards/meter/std": 0.01168668083846569, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9875234365463257, "rewards/total_composite/std": 0.01168668083846569, "reward": 0.9875234365463257, "reward_std": 0.01168669294565916, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.054695580154657364, "sampling/sampling_logp_difference/max": 1.2924197912216187, "sampling/importance_sampling_ratio/min": 0.27460548281669617, "sampling/importance_sampling_ratio/mean": 1.0194075107574463, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5414317175745964, "clip_ratio/low_mean": 0.009980237111449242, "clip_ratio/low_min": 0.009980237111449242, "clip_ratio/high_mean": 0.03236477100290358, "clip_ratio/high_max": 0.03236477100290358, "clip_ratio/region_mean": 0.04234500811435282, "reward_total_mean": 0.9875234365463257, "reward_meter_mean": 0.9875234365463257, "reward_meter_std": 0.01168668083846569, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9875234365463257, "reward_total_composite_std": 0.01168668083846569} {"timestamp_utc": "2026-04-12T00:59:01Z", "mode": "train", "global_step": 1855, "epoch": 0.0745069687110897, "loss": 0.0017, "grad_norm": 2.625075101852417, "learning_rate": 4.381818181818182e-06, "num_tokens": 4178031.0, "completions/mean_length": 56.0, "completions/min_length": 56.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.0, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.9880209565162659, "rewards/meter/std": 0.007883170619606972, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9880209565162659, "rewards/total_composite/std": 0.007883170619606972, "reward": 0.9880209565162659, "reward_std": 0.0078831622377038, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010663346387445927, "sampling/sampling_logp_difference/max": 0.7668309211730957, "sampling/importance_sampling_ratio/min": 0.46448272466659546, "sampling/importance_sampling_ratio/mean": 1.0048121213912964, "sampling/importance_sampling_ratio/max": 1.396715521812439, "entropy": 0.0864438284188509, "clip_ratio/low_mean": 0.0022321429569274187, "clip_ratio/low_min": 0.0022321429569274187, "clip_ratio/high_mean": 0.008928571827709675, "clip_ratio/high_max": 0.008928571827709675, "clip_ratio/region_mean": 0.011160714784637094, "reward_total_mean": 0.9880209565162659, "reward_meter_mean": 0.9880209565162659, "reward_meter_std": 0.007883170619606972, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9880209565162659, "reward_total_composite_std": 0.007883170619606972} {"timestamp_utc": "2026-04-12T00:59:06Z", "mode": "train", "global_step": 1856, "epoch": 0.07454713419287465, "loss": 0.0108, "grad_norm": 3.003889799118042, "learning_rate": 4.378787878787879e-06, "num_tokens": 4180182.0, "completions/mean_length": 111.875, "completions/min_length": 109.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 111.875, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.9974768757820129, "rewards/meter/std": 0.0010170461609959602, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.8977147340774536, "rewards/total_composite/std": 0.10649988800287247, "reward": 0.8977147340774536, "reward_std": 0.10649988800287247, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04684070870280266, "sampling/sampling_logp_difference/max": 1.5592315196990967, "sampling/importance_sampling_ratio/min": 0.2102976143360138, "sampling/importance_sampling_ratio/mean": 1.004133701324463, "sampling/importance_sampling_ratio/max": 1.5614416599273682, "entropy": 0.35822908766567707, "clip_ratio/low_mean": 0.007754115387797356, "clip_ratio/low_min": 0.007754115387797356, "clip_ratio/high_mean": 0.023573311744257808, "clip_ratio/high_max": 0.023573311744257808, "clip_ratio/region_mean": 0.03132742713205516, "reward_total_mean": 0.8977147340774536, "reward_meter_mean": 0.9974768757820129, "reward_meter_std": 0.0010170461609959602, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_total_composite_mean": 0.8977147340774536, "reward_total_composite_std": 0.10649988800287247} {"timestamp_utc": "2026-04-12T00:59:11Z", "mode": "train", "global_step": 1857, "epoch": 0.0745872996746596, "loss": -0.0105, "grad_norm": 1.938165307044983, "learning_rate": 4.375757575757576e-06, "num_tokens": 4181844.0, "completions/mean_length": 56.75, "completions/min_length": 56.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.75, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9916501641273499, "rewards/meter/std": 0.001931402483023703, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9916501641273499, "rewards/total_composite/std": 0.001931402483023703, "reward": 0.9916501641273499, "reward_std": 0.0019314087694510818, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013949423097074032, "sampling/sampling_logp_difference/max": 1.0583875179290771, "sampling/importance_sampling_ratio/min": 0.3470149040222168, "sampling/importance_sampling_ratio/mean": 1.0090504884719849, "sampling/importance_sampling_ratio/max": 1.4867905378341675, "entropy": 0.09877756051719189, "clip_ratio/low_mean": 0.017778822453692555, "clip_ratio/low_min": 0.017778822453692555, "clip_ratio/high_mean": 0.0021551724057644606, "clip_ratio/high_max": 0.0021551724057644606, "clip_ratio/region_mean": 0.019933994859457016, "reward_total_mean": 0.9916501641273499, "reward_meter_mean": 0.9916501641273499, "reward_meter_std": 0.001931402483023703, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9916501641273499, "reward_total_composite_std": 0.001931402483023703} {"timestamp_utc": "2026-04-12T00:59:16Z", "mode": "train", "global_step": 1858, "epoch": 0.07462746515644456, "loss": -0.0057, "grad_norm": 2.222642421722412, "learning_rate": 4.372727272727273e-06, "num_tokens": 4183958.0, "completions/mean_length": 88.25, "completions/min_length": 82.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 88.25, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9945516586303711, "rewards/meter/std": 0.0012705601984634995, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9945516586303711, "rewards/total_composite/std": 0.0012705601984634995, "reward": 0.9945516586303711, "reward_std": 0.0012705654371529818, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012971967458724976, "sampling/sampling_logp_difference/max": 0.851701021194458, "sampling/importance_sampling_ratio/min": 0.4266884922981262, "sampling/importance_sampling_ratio/mean": 1.0038167238235474, "sampling/importance_sampling_ratio/max": 1.6056901216506958, "entropy": 0.09635706804692745, "clip_ratio/low_mean": 0.0028572361916303635, "clip_ratio/low_min": 0.0028572361916303635, "clip_ratio/high_mean": 0.007095551351085305, "clip_ratio/high_max": 0.007095551351085305, "clip_ratio/region_mean": 0.009952787542715669, "reward_total_mean": 0.9945516586303711, "reward_meter_mean": 0.9945516586303711, "reward_meter_std": 0.0012705601984634995, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9945516586303711, "reward_total_composite_std": 0.0012705601984634995} {"timestamp_utc": "2026-04-12T00:59:20Z", "mode": "train", "global_step": 1859, "epoch": 0.07466763063822951, "loss": -0.0001, "grad_norm": 4.79885721206665, "learning_rate": 4.36969696969697e-06, "num_tokens": 4185702.0, "completions/mean_length": 51.0, "completions/min_length": 51.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.0, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.9259762763977051, "rewards/meter/std": 0.0015891619259491563, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9259762763977051, "rewards/total_composite/std": 0.0015891619259491563, "reward": 0.9259762763977051, "reward_std": 0.001589166815392673, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012182224541902542, "sampling/sampling_logp_difference/max": 0.820074200630188, "sampling/importance_sampling_ratio/min": 0.4403989613056183, "sampling/importance_sampling_ratio/mean": 0.9998874068260193, "sampling/importance_sampling_ratio/max": 1.6099872589111328, "entropy": 0.049468324054032564, "clip_ratio/low_mean": 0.007352941436693072, "clip_ratio/low_min": 0.007352941436693072, "clip_ratio/high_mean": 0.0049019609577953815, "clip_ratio/high_max": 0.0049019609577953815, "clip_ratio/region_mean": 0.012254902394488454, "reward_total_mean": 0.9259762763977051, "reward_meter_mean": 0.9259762763977051, "reward_meter_std": 0.0015891619259491563, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9259762763977051, "reward_total_composite_std": 0.0015891619259491563} {"timestamp_utc": "2026-04-12T00:59:26Z", "mode": "train", "global_step": 1860, "epoch": 0.07470779612001446, "loss": -0.0032, "grad_norm": 2.1606459617614746, "learning_rate": 4.366666666666667e-06, "num_tokens": 4188182.0, "completions/mean_length": 135.0, "completions/min_length": 132.0, "completions/max_length": 138.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 135.0, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.989547610282898, "rewards/meter/std": 0.007958509027957916, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8928571939468384, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.8834315538406372, "rewards/total_composite/std": 0.06423989683389664, "reward": 0.8834315538406372, "reward_std": 0.06423989683389664, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01665782928466797, "sampling/sampling_logp_difference/max": 1.199911117553711, "sampling/importance_sampling_ratio/min": 0.30122095346450806, "sampling/importance_sampling_ratio/mean": 1.0032159090042114, "sampling/importance_sampling_ratio/max": 1.8788100481033325, "entropy": 0.100110930390656, "clip_ratio/low_mean": 0.008354876074008644, "clip_ratio/low_min": 0.008354876074008644, "clip_ratio/high_mean": 0.006407182663679123, "clip_ratio/high_max": 0.006407182663679123, "clip_ratio/region_mean": 0.014762058737687767, "reward_total_mean": 0.8834315538406372, "reward_meter_mean": 0.989547610282898, "reward_meter_std": 0.007958509027957916, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8928571939468384, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.8834315538406372, "reward_total_composite_std": 0.06423989683389664} {"timestamp_utc": "2026-04-12T00:59:31Z", "mode": "train", "global_step": 1861, "epoch": 0.07474796160179942, "loss": 0.0208, "grad_norm": 2.7288973331451416, "learning_rate": 4.363636363636364e-06, "num_tokens": 4190278.0, "completions/mean_length": 101.0, "completions/min_length": 97.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.0, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.7360309958457947, "rewards/meter/std": 0.27524590492248535, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.7065380811691284, "rewards/total_composite/std": 0.27462705969810486, "reward": 0.7065380811691284, "reward_std": 0.27462708950042725, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0317775197327137, "sampling/sampling_logp_difference/max": 2.5332324504852295, "sampling/importance_sampling_ratio/min": 0.07940194010734558, "sampling/importance_sampling_ratio/mean": 1.0014276504516602, "sampling/importance_sampling_ratio/max": 1.5228219032287598, "entropy": 0.16979045886546373, "clip_ratio/low_mean": 0.009549640817567706, "clip_ratio/low_min": 0.009549640817567706, "clip_ratio/high_mean": 0.0124762476189062, "clip_ratio/high_max": 0.0124762476189062, "clip_ratio/region_mean": 0.022025888436473906, "reward_total_mean": 0.7065380811691284, "reward_meter_mean": 0.7360309958457947, "reward_meter_std": 0.27524590492248535, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.7065380811691284, "reward_total_composite_std": 0.27462705969810486} {"timestamp_utc": "2026-04-12T00:59:37Z", "mode": "train", "global_step": 1862, "epoch": 0.07478812708358437, "loss": 0.0047, "grad_norm": 1.0823556184768677, "learning_rate": 4.36060606060606e-06, "num_tokens": 4193661.0, "completions/mean_length": 221.875, "completions/min_length": 221.0, "completions/max_length": 225.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 221.875, "completions/min_terminated_length": 221.0, "completions/max_terminated_length": 225.0, "rewards/meter/mean": 0.9964575171470642, "rewards/meter/std": 0.0009934601839631796, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6643049716949463, "rewards/total_composite/std": 0.0006622962537221611, "reward": 0.6643049716949463, "reward_std": 0.000662296952214092, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01341662835329771, "sampling/sampling_logp_difference/max": 9.43514347076416, "sampling/importance_sampling_ratio/min": 7.986734999576584e-05, "sampling/importance_sampling_ratio/mean": 1.0002436637878418, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03992933081462979, "clip_ratio/low_mean": 0.0011111111380159855, "clip_ratio/low_min": 0.0011111111380159855, "clip_ratio/high_mean": 0.005077758920378983, "clip_ratio/high_max": 0.005077758920378983, "clip_ratio/region_mean": 0.0061888700583949685, "reward_total_mean": 0.6643049716949463, "reward_meter_mean": 0.9964575171470642, "reward_meter_std": 0.0009934601839631796, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6643049716949463, "reward_total_composite_std": 0.0006622962537221611} {"timestamp_utc": "2026-04-12T00:59:42Z", "mode": "train", "global_step": 1863, "epoch": 0.07482829256536933, "loss": 0.0087, "grad_norm": 5.034853458404541, "learning_rate": 4.3575757575757576e-06, "num_tokens": 4195615.0, "completions/mean_length": 73.25, "completions/min_length": 67.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.25, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9600696563720703, "rewards/meter/std": 0.09926661849021912, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9600696563720703, "rewards/total_composite/std": 0.09926661849021912, "reward": 0.9600696563720703, "reward_std": 0.09926661849021912, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07004112005233765, "sampling/sampling_logp_difference/max": 2.483992576599121, "sampling/importance_sampling_ratio/min": 0.08340954035520554, "sampling/importance_sampling_ratio/mean": 1.0097596645355225, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6060812920331955, "clip_ratio/low_mean": 0.006666666828095913, "clip_ratio/low_min": 0.006666666828095913, "clip_ratio/high_mean": 0.05100303632207215, "clip_ratio/high_max": 0.05100303632207215, "clip_ratio/region_mean": 0.05766970315016806, "reward_total_mean": 0.9600696563720703, "reward_meter_mean": 0.9600696563720703, "reward_meter_std": 0.09926661849021912, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9600696563720703, "reward_total_composite_std": 0.09926661849021912} {"timestamp_utc": "2026-04-12T00:59:47Z", "mode": "train", "global_step": 1864, "epoch": 0.07486845804715428, "loss": -0.0061, "grad_norm": 1.9747538566589355, "learning_rate": 4.354545454545455e-06, "num_tokens": 4197677.0, "completions/mean_length": 89.75, "completions/min_length": 88.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.75, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9951763153076172, "rewards/meter/std": 0.0005524771986529231, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9951763153076172, "rewards/total_composite/std": 0.0005524771986529231, "reward": 0.9951763153076172, "reward_std": 0.0005524645675905049, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008235161192715168, "sampling/sampling_logp_difference/max": 1.078155517578125, "sampling/importance_sampling_ratio/min": 0.34022247791290283, "sampling/importance_sampling_ratio/mean": 1.0047131776809692, "sampling/importance_sampling_ratio/max": 1.4608083963394165, "entropy": 0.05834774626418948, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0013888889225199819, "clip_ratio/high_max": 0.0013888889225199819, "clip_ratio/region_mean": 0.0013888889225199819, "reward_total_mean": 0.9951763153076172, "reward_meter_mean": 0.9951763153076172, "reward_meter_std": 0.0005524771986529231, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9951763153076172, "reward_total_composite_std": 0.0005524771986529231} {"timestamp_utc": "2026-04-12T00:59:51Z", "mode": "train", "global_step": 1865, "epoch": 0.07490862352893923, "loss": 0.0111, "grad_norm": 3.9621639251708984, "learning_rate": 4.351515151515152e-06, "num_tokens": 4199512.0, "completions/mean_length": 69.375, "completions/min_length": 69.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.375, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9972849488258362, "rewards/meter/std": 0.0004286824550945312, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9972849488258362, "rewards/total_composite/std": 0.0004286824550945312, "reward": 0.9972849488258362, "reward_std": 0.000428676517913118, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015396162867546082, "sampling/sampling_logp_difference/max": 0.8150291442871094, "sampling/importance_sampling_ratio/min": 0.4426264464855194, "sampling/importance_sampling_ratio/mean": 1.003419280052185, "sampling/importance_sampling_ratio/max": 1.5455111265182495, "entropy": 0.12075466383248568, "clip_ratio/low_mean": 0.010716472752392292, "clip_ratio/low_min": 0.010716472752392292, "clip_ratio/high_mean": 0.0036231884732842445, "clip_ratio/high_max": 0.0036231884732842445, "clip_ratio/region_mean": 0.014339661225676537, "reward_total_mean": 0.9972849488258362, "reward_meter_mean": 0.9972849488258362, "reward_meter_std": 0.0004286824550945312, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9972849488258362, "reward_total_composite_std": 0.0004286824550945312} {"timestamp_utc": "2026-04-12T00:59:56Z", "mode": "train", "global_step": 1866, "epoch": 0.07494878901072419, "loss": 0.0048, "grad_norm": 1.6598762273788452, "learning_rate": 4.348484848484849e-06, "num_tokens": 4201587.0, "completions/mean_length": 103.375, "completions/min_length": 103.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.375, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.9974133968353271, "rewards/meter/std": 0.00014544985606335104, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9974133968353271, "rewards/total_composite/std": 0.00014544985606335104, "reward": 0.9974133968353271, "reward_std": 0.00014544471923727542, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009199890308082104, "sampling/sampling_logp_difference/max": 0.9815373420715332, "sampling/importance_sampling_ratio/min": 0.3747345805168152, "sampling/importance_sampling_ratio/mean": 0.9997120499610901, "sampling/importance_sampling_ratio/max": 1.4277647733688354, "entropy": 0.03981838142499328, "clip_ratio/low_mean": 0.0036407767329365015, "clip_ratio/low_min": 0.0036407767329365015, "clip_ratio/high_mean": 0.0036407767329365015, "clip_ratio/high_max": 0.0036407767329365015, "clip_ratio/region_mean": 0.007281553465873003, "reward_total_mean": 0.9974133968353271, "reward_meter_mean": 0.9974133968353271, "reward_meter_std": 0.00014544985606335104, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9974133968353271, "reward_total_composite_std": 0.00014544985606335104} {"timestamp_utc": "2026-04-12T01:00:05Z", "mode": "train", "global_step": 1867, "epoch": 0.07498895449250914, "loss": 0.0447, "grad_norm": 2.981818675994873, "learning_rate": 4.345454545454546e-06, "num_tokens": 4206088.0, "completions/mean_length": 356.625, "completions/min_length": 331.0, "completions/max_length": 391.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 356.625, "completions/min_terminated_length": 331.0, "completions/max_terminated_length": 391.0, "rewards/meter/mean": 0.904925525188446, "rewards/meter/std": 0.24405907094478607, "rewards/count_adherence/mean": 0.6333333253860474, "rewards/count_adherence/std": 0.035634830594062805, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8377193212509155, "rewards/repeat_penalty/std": 0.11382605135440826, "rewards/total_composite/mean": 0.46806132793426514, "rewards/total_composite/std": 0.1208736002445221, "reward": 0.46806132793426514, "reward_std": 0.1208736002445221, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05986325070261955, "sampling/sampling_logp_difference/max": 14.691458702087402, "sampling/importance_sampling_ratio/min": 4.1646697468422644e-07, "sampling/importance_sampling_ratio/mean": 1.0076172351837158, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6004703845828772, "clip_ratio/low_mean": 0.01117079914547503, "clip_ratio/low_min": 0.01117079914547503, "clip_ratio/high_mean": 0.01905520213767886, "clip_ratio/high_max": 0.01905520213767886, "clip_ratio/region_mean": 0.03022600128315389, "reward_total_mean": 0.46806132793426514, "reward_meter_mean": 0.904925525188446, "reward_meter_std": 0.24405907094478607, "reward_count_adherence_mean": 0.6333333253860474, "reward_count_adherence_std": 0.035634830594062805, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8377193212509155, "reward_repeat_penalty_std": 0.11382605135440826, "reward_total_composite_mean": 0.46806132793426514, "reward_total_composite_std": 0.1208736002445221} {"timestamp_utc": "2026-04-12T01:00:10Z", "mode": "train", "global_step": 1868, "epoch": 0.0750291199742941, "loss": -0.0004, "grad_norm": 0.8906328678131104, "learning_rate": 4.342424242424243e-06, "num_tokens": 4208151.0, "completions/mean_length": 89.875, "completions/min_length": 88.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.875, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9952363967895508, "rewards/meter/std": 0.00016089908604044467, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9952363967895508, "rewards/total_composite/std": 0.00016089908604044467, "reward": 0.9952363967895508, "reward_std": 0.0001608873571967706, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009152467362582684, "sampling/sampling_logp_difference/max": 0.7741870880126953, "sampling/importance_sampling_ratio/min": 0.4610784351825714, "sampling/importance_sampling_ratio/mean": 1.002822756767273, "sampling/importance_sampling_ratio/max": 1.6893128156661987, "entropy": 0.06600452493876219, "clip_ratio/low_mean": 0.0028093435103073716, "clip_ratio/low_min": 0.0028093435103073716, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0028093435103073716, "reward_total_mean": 0.9952363967895508, "reward_meter_mean": 0.9952363967895508, "reward_meter_std": 0.00016089908604044467, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9952363967895508, "reward_total_composite_std": 0.00016089908604044467} {"timestamp_utc": "2026-04-12T01:00:15Z", "mode": "train", "global_step": 1869, "epoch": 0.07506928545607905, "loss": 0.0019, "grad_norm": 2.3382408618927, "learning_rate": 4.33939393939394e-06, "num_tokens": 4210238.0, "completions/mean_length": 102.875, "completions/min_length": 102.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.875, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.9974008798599243, "rewards/meter/std": 0.00030872138449922204, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9974008798599243, "rewards/total_composite/std": 0.00030872138449922204, "reward": 0.9974008798599243, "reward_std": 0.00030873037758283317, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009639319032430649, "sampling/sampling_logp_difference/max": 1.461493968963623, "sampling/importance_sampling_ratio/min": 0.23188959062099457, "sampling/importance_sampling_ratio/mean": 0.9990761280059814, "sampling/importance_sampling_ratio/max": 1.2280147075653076, "entropy": 0.03081246791407466, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0012135922443121672, "clip_ratio/high_max": 0.0012135922443121672, "clip_ratio/region_mean": 0.0012135922443121672, "reward_total_mean": 0.9974008798599243, "reward_meter_mean": 0.9974008798599243, "reward_meter_std": 0.00030872138449922204, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9974008798599243, "reward_total_composite_std": 0.00030872138449922204} {"timestamp_utc": "2026-04-12T01:00:19Z", "mode": "train", "global_step": 1870, "epoch": 0.075109450937864, "loss": -0.0106, "grad_norm": 4.046680927276611, "learning_rate": 4.336363636363637e-06, "num_tokens": 4212193.0, "completions/mean_length": 76.375, "completions/min_length": 70.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.375, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9944717288017273, "rewards/meter/std": 0.003807853441685438, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9944717288017273, "rewards/total_composite/std": 0.003807853441685438, "reward": 0.9944717288017273, "reward_std": 0.003807859495282173, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04645724594593048, "sampling/sampling_logp_difference/max": 1.8717412948608398, "sampling/importance_sampling_ratio/min": 0.15385551750659943, "sampling/importance_sampling_ratio/mean": 1.011504054069519, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4157518967986107, "clip_ratio/low_mean": 0.02176366886124015, "clip_ratio/low_min": 0.02176366886124015, "clip_ratio/high_mean": 0.011222697328776121, "clip_ratio/high_max": 0.011222697328776121, "clip_ratio/region_mean": 0.03298636619001627, "reward_total_mean": 0.9944717288017273, "reward_meter_mean": 0.9944717288017273, "reward_meter_std": 0.003807853441685438, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9944717288017273, "reward_total_composite_std": 0.003807853441685438} {"timestamp_utc": "2026-04-12T01:00:24Z", "mode": "train", "global_step": 1871, "epoch": 0.07514961641964896, "loss": -0.0038, "grad_norm": 2.565702438354492, "learning_rate": 4.333333333333334e-06, "num_tokens": 4213938.0, "completions/mean_length": 59.125, "completions/min_length": 58.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.125, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9952658414840698, "rewards/meter/std": 0.0007007194799371064, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9952658414840698, "rewards/total_composite/std": 0.0007007194799371064, "reward": 0.9952658414840698, "reward_std": 0.000700704287737608, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013889300636947155, "sampling/sampling_logp_difference/max": 0.7738056182861328, "sampling/importance_sampling_ratio/min": 0.461254358291626, "sampling/importance_sampling_ratio/mean": 0.996799886226654, "sampling/importance_sampling_ratio/max": 1.3448500633239746, "entropy": 0.057940175756812096, "clip_ratio/low_mean": 0.0021186440717428923, "clip_ratio/low_min": 0.0021186440717428923, "clip_ratio/high_mean": 0.004167824285104871, "clip_ratio/high_max": 0.004167824285104871, "clip_ratio/region_mean": 0.006286468356847763, "reward_total_mean": 0.9952658414840698, "reward_meter_mean": 0.9952658414840698, "reward_meter_std": 0.0007007194799371064, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9952658414840698, "reward_total_composite_std": 0.0007007194799371064} {"timestamp_utc": "2026-04-12T01:00:34Z", "mode": "train", "global_step": 1872, "epoch": 0.07518978190143391, "loss": -0.3047, "grad_norm": 1.4247676134109497, "learning_rate": 4.330303030303031e-06, "num_tokens": 4216745.0, "completions/mean_length": 294.875, "completions/min_length": 212.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 222.5, "completions/min_terminated_length": 212.0, "completions/max_terminated_length": 233.0, "rewards/meter/mean": 0.8686028718948364, "rewards/meter/std": 0.3345178961753845, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 0.9460227489471436, "rewards/repeat_penalty/std": 0.06318090111017227, "rewards/total_composite/mean": 0.6971731185913086, "rewards/total_composite/std": 0.433406800031662, "reward": 0.6971731185913086, "reward_std": 0.433406800031662, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0717155784368515, "sampling/sampling_logp_difference/max": 1.2843422889709473, "sampling/importance_sampling_ratio/min": 0.27683258056640625, "sampling/importance_sampling_ratio/mean": 1.018416404724121, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5619211308658123, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.03716663923114538, "clip_ratio/high_max": 0.03716663923114538, "clip_ratio/region_mean": 0.03716663923114538, "reward_total_mean": 0.6971731185913086, "reward_meter_mean": 0.8686028718948364, "reward_meter_std": 0.3345178961753845, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 0.9460227489471436, "reward_repeat_penalty_std": 0.06318090111017227, "reward_total_composite_mean": 0.6971731185913086, "reward_total_composite_std": 0.433406800031662} {"timestamp_utc": "2026-04-12T01:00:43Z", "mode": "train", "global_step": 1873, "epoch": 0.07522994738321886, "loss": -0.2461, "grad_norm": 1.2419461011886597, "learning_rate": 4.327272727272728e-06, "num_tokens": 4219511.0, "completions/mean_length": 235.75, "completions/min_length": 190.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 196.2857208251953, "completions/min_terminated_length": 190.0, "completions/max_terminated_length": 203.0, "rewards/meter/mean": 0.8880919814109802, "rewards/meter/std": 0.2659795582294464, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9611110687255859, "rewards/repeat_penalty/std": 0.07582584023475647, "rewards/total_composite/mean": 0.851571798324585, "rewards/total_composite/std": 0.26690804958343506, "reward": 0.851571798324585, "reward_std": 0.26690801978111267, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.058800045400857925, "sampling/sampling_logp_difference/max": 1.3028392791748047, "sampling/importance_sampling_ratio/min": 0.2717590928077698, "sampling/importance_sampling_ratio/mean": 1.0172072649002075, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5523100271821022, "clip_ratio/low_mean": 0.0032552082557231188, "clip_ratio/low_min": 0.0032552082557231188, "clip_ratio/high_mean": 0.02799722831696272, "clip_ratio/high_max": 0.02799722831696272, "clip_ratio/region_mean": 0.03125243657268584, "reward_total_mean": 0.851571798324585, "reward_meter_mean": 0.8880919814109802, "reward_meter_std": 0.2659795582294464, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9611110687255859, "reward_repeat_penalty_std": 0.07582584023475647, "reward_total_composite_mean": 0.851571798324585, "reward_total_composite_std": 0.26690804958343506} {"timestamp_utc": "2026-04-12T01:00:49Z", "mode": "train", "global_step": 1874, "epoch": 0.07527011286500382, "loss": -0.0071, "grad_norm": 6.65327787399292, "learning_rate": 4.324242424242425e-06, "num_tokens": 4221629.0, "completions/mean_length": 105.75, "completions/min_length": 99.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.75, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.9625563621520996, "rewards/meter/std": 0.03560182824730873, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.9133010506629944, "rewards/total_composite/std": 0.08394214510917664, "reward": 0.9133010506629944, "reward_std": 0.08394213020801544, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06265782564878464, "sampling/sampling_logp_difference/max": 1.7746086120605469, "sampling/importance_sampling_ratio/min": 0.16954979300498962, "sampling/importance_sampling_ratio/mean": 0.9978815317153931, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40878177247941494, "clip_ratio/low_mean": 0.012383450288325548, "clip_ratio/low_min": 0.012383450288325548, "clip_ratio/high_mean": 0.0333196020219475, "clip_ratio/high_max": 0.0333196020219475, "clip_ratio/region_mean": 0.04570305231027305, "reward_total_mean": 0.9133010506629944, "reward_meter_mean": 0.9625563621520996, "reward_meter_std": 0.03560182824730873, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.9133010506629944, "reward_total_composite_std": 0.08394214510917664} {"timestamp_utc": "2026-04-12T01:00:55Z", "mode": "train", "global_step": 1875, "epoch": 0.07531027834678877, "loss": 0.0297, "grad_norm": 3.1858253479003906, "learning_rate": 4.321212121212121e-06, "num_tokens": 4224968.0, "completions/mean_length": 202.375, "completions/min_length": 173.0, "completions/max_length": 221.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 202.375, "completions/min_terminated_length": 173.0, "completions/max_terminated_length": 221.0, "rewards/meter/mean": 0.888024091720581, "rewards/meter/std": 0.2775907516479492, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9608585834503174, "rewards/repeat_penalty/std": 0.08052574098110199, "rewards/total_composite/mean": 0.8535768985748291, "rewards/total_composite/std": 0.29564857482910156, "reward": 0.8535768985748291, "reward_std": 0.29564857482910156, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05819166079163551, "sampling/sampling_logp_difference/max": 1.7184324264526367, "sampling/importance_sampling_ratio/min": 0.17934706807136536, "sampling/importance_sampling_ratio/mean": 1.0147325992584229, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6360356472432613, "clip_ratio/low_mean": 0.008985401596873999, "clip_ratio/low_min": 0.008985401596873999, "clip_ratio/high_mean": 0.02583647519350052, "clip_ratio/high_max": 0.02583647519350052, "clip_ratio/region_mean": 0.03482187679037452, "reward_total_mean": 0.8535768985748291, "reward_meter_mean": 0.888024091720581, "reward_meter_std": 0.2775907516479492, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9608585834503174, "reward_repeat_penalty_std": 0.08052574098110199, "reward_total_composite_mean": 0.8535768985748291, "reward_total_composite_std": 0.29564857482910156} {"timestamp_utc": "2026-04-12T01:01:00Z", "mode": "train", "global_step": 1876, "epoch": 0.07535044382857373, "loss": 0.0171, "grad_norm": 5.3259077072143555, "learning_rate": 4.3181818181818185e-06, "num_tokens": 4227160.0, "completions/mean_length": 72.0, "completions/min_length": 69.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.0, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.919043779373169, "rewards/meter/std": 0.19243919849395752, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.919043779373169, "rewards/total_composite/std": 0.19243919849395752, "reward": 0.919043779373169, "reward_std": 0.19243919849395752, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04635407403111458, "sampling/sampling_logp_difference/max": 1.1690726280212402, "sampling/importance_sampling_ratio/min": 0.310654878616333, "sampling/importance_sampling_ratio/mean": 1.0002164840698242, "sampling/importance_sampling_ratio/max": 1.5501129627227783, "entropy": 0.309705201536417, "clip_ratio/low_mean": 0.010135134682059288, "clip_ratio/low_min": 0.010135134682059288, "clip_ratio/high_mean": 0.034461831324733794, "clip_ratio/high_max": 0.034461831324733794, "clip_ratio/region_mean": 0.04459696600679308, "reward_total_mean": 0.919043779373169, "reward_meter_mean": 0.919043779373169, "reward_meter_std": 0.19243919849395752, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.919043779373169, "reward_total_composite_std": 0.19243919849395752} {"timestamp_utc": "2026-04-12T01:01:06Z", "mode": "train", "global_step": 1877, "epoch": 0.07539060931035868, "loss": -0.0123, "grad_norm": 2.582993745803833, "learning_rate": 4.315151515151516e-06, "num_tokens": 4230422.0, "completions/mean_length": 204.75, "completions/min_length": 192.0, "completions/max_length": 227.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 204.75, "completions/min_terminated_length": 192.0, "completions/max_terminated_length": 227.0, "rewards/meter/mean": 0.9647862911224365, "rewards/meter/std": 0.05925486981868744, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9597222208976746, "rewards/repeat_penalty/std": 0.055694278329610825, "rewards/total_composite/mean": 0.9254087209701538, "rewards/total_composite/std": 0.07272467762231827, "reward": 0.9254087209701538, "reward_std": 0.07272466272115707, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05267081782221794, "sampling/sampling_logp_difference/max": 1.6134209632873535, "sampling/importance_sampling_ratio/min": 0.1992049664258957, "sampling/importance_sampling_ratio/mean": 1.0190201997756958, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5658724009990692, "clip_ratio/low_mean": 0.015004753833636642, "clip_ratio/low_min": 0.015004753833636642, "clip_ratio/high_mean": 0.015678069554269314, "clip_ratio/high_max": 0.015678069554269314, "clip_ratio/region_mean": 0.030682823387905955, "reward_total_mean": 0.9254087209701538, "reward_meter_mean": 0.9647862911224365, "reward_meter_std": 0.05925486981868744, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9597222208976746, "reward_repeat_penalty_std": 0.055694278329610825, "reward_total_composite_mean": 0.9254087209701538, "reward_total_composite_std": 0.07272467762231827} {"timestamp_utc": "2026-04-12T01:01:16Z", "mode": "train", "global_step": 1878, "epoch": 0.07543077479214363, "loss": -0.2576, "grad_norm": 1.26083242893219, "learning_rate": 4.312121212121212e-06, "num_tokens": 4234422.0, "completions/mean_length": 332.0, "completions/min_length": 285.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 306.2857360839844, "completions/min_terminated_length": 285.0, "completions/max_terminated_length": 319.0, "rewards/meter/mean": 0.8925167322158813, "rewards/meter/std": 0.15122567117214203, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.05345224589109421, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.919905424118042, "rewards/repeat_penalty/std": 0.1122746393084526, "rewards/total_composite/mean": 0.5566210746765137, "rewards/total_composite/std": 0.25293970108032227, "reward": 0.5566210746765137, "reward_std": 0.25293970108032227, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07019904255867004, "sampling/sampling_logp_difference/max": 14.930951118469238, "sampling/importance_sampling_ratio/min": 3.277708344739949e-07, "sampling/importance_sampling_ratio/mean": 1.012736439704895, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.48640119284391403, "clip_ratio/low_mean": 0.011000885395333171, "clip_ratio/low_min": 0.011000885395333171, "clip_ratio/high_mean": 0.032999717397615314, "clip_ratio/high_max": 0.032999717397615314, "clip_ratio/region_mean": 0.044000602792948484, "reward_total_mean": 0.5566210746765137, "reward_meter_mean": 0.8925167322158813, "reward_meter_std": 0.15122567117214203, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.05345224589109421, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.919905424118042, "reward_repeat_penalty_std": 0.1122746393084526, "reward_total_composite_mean": 0.5566210746765137, "reward_total_composite_std": 0.25293970108032227} {"timestamp_utc": "2026-04-12T01:01:26Z", "mode": "train", "global_step": 1879, "epoch": 0.07547094027392859, "loss": -0.2401, "grad_norm": 0.8991881012916565, "learning_rate": 4.309090909090909e-06, "num_tokens": 4236988.0, "completions/mean_length": 199.75, "completions/min_length": 146.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 155.1428680419922, "completions/min_terminated_length": 146.0, "completions/max_terminated_length": 159.0, "rewards/meter/mean": 0.8689680099487305, "rewards/meter/std": 0.3431207239627838, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.866482138633728, "rewards/total_composite/std": 0.3501509428024292, "reward": 0.866482138633728, "reward_std": 0.3501509428024292, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.052908286452293396, "sampling/sampling_logp_difference/max": 1.6577634811401367, "sampling/importance_sampling_ratio/min": 0.19056470692157745, "sampling/importance_sampling_ratio/mean": 1.0059314966201782, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3837529495358467, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0337692714529112, "clip_ratio/high_max": 0.0337692714529112, "clip_ratio/region_mean": 0.0337692714529112, "reward_total_mean": 0.866482138633728, "reward_meter_mean": 0.8689680099487305, "reward_meter_std": 0.3431207239627838, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.866482138633728, "reward_total_composite_std": 0.3501509428024292} {"timestamp_utc": "2026-04-12T01:01:31Z", "mode": "train", "global_step": 1880, "epoch": 0.07551110575571354, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.306060606060607e-06, "num_tokens": 4238773.0, "completions/mean_length": 69.125, "completions/min_length": 69.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.125, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.997582197189331, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.997582197189331, "rewards/total_composite/std": 0.0, "reward": 0.997582197189331, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0021389967296272516, "sampling/sampling_logp_difference/max": 0.24272990226745605, "sampling/importance_sampling_ratio/min": 0.7844833731651306, "sampling/importance_sampling_ratio/mean": 1.000409722328186, "sampling/importance_sampling_ratio/max": 1.1089041233062744, "entropy": 0.013600267469882965, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.997582197189331, "reward_meter_mean": 0.997582197189331, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.997582197189331, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:01:35Z", "mode": "train", "global_step": 1881, "epoch": 0.0755512712374985, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.303030303030303e-06, "num_tokens": 4240565.0, "completions/mean_length": 69.0, "completions/min_length": 69.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.0, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.997582197189331, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.997582197189331, "rewards/total_composite/std": 0.0, "reward": 0.997582197189331, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.001675679231993854, "sampling/sampling_logp_difference/max": 0.10361464321613312, "sampling/importance_sampling_ratio/min": 0.9542538523674011, "sampling/importance_sampling_ratio/mean": 1.0012025833129883, "sampling/importance_sampling_ratio/max": 1.1091729402542114, "entropy": 0.014390837168321013, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.997582197189331, "reward_meter_mean": 0.997582197189331, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.997582197189331, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:01:40Z", "mode": "train", "global_step": 1882, "epoch": 0.07559143671928345, "loss": 0.0199, "grad_norm": 2.6807198524475098, "learning_rate": 4.3e-06, "num_tokens": 4242138.0, "completions/mean_length": 42.625, "completions/min_length": 40.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.625, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.9950450658798218, "rewards/meter/std": 0.00384511542506516, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9950450658798218, "rewards/total_composite/std": 0.00384511542506516, "reward": 0.9950450658798218, "reward_std": 0.0038451338186860085, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03229609876871109, "sampling/sampling_logp_difference/max": 0.9218476414680481, "sampling/importance_sampling_ratio/min": 0.39778339862823486, "sampling/importance_sampling_ratio/mean": 1.0034741163253784, "sampling/importance_sampling_ratio/max": 1.5244181156158447, "entropy": 0.2804165482521057, "clip_ratio/low_mean": 0.005681818351149559, "clip_ratio/low_min": 0.005681818351149559, "clip_ratio/high_mean": 0.0058139534667134285, "clip_ratio/high_max": 0.0058139534667134285, "clip_ratio/region_mean": 0.011495771817862988, "reward_total_mean": 0.9950450658798218, "reward_meter_mean": 0.9950450658798218, "reward_meter_std": 0.00384511542506516, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9950450658798218, "reward_total_composite_std": 0.00384511542506516} {"timestamp_utc": "2026-04-12T01:01:45Z", "mode": "train", "global_step": 1883, "epoch": 0.0756316022010684, "loss": -0.0012, "grad_norm": 2.456183433532715, "learning_rate": 4.296969696969698e-06, "num_tokens": 4244703.0, "completions/mean_length": 160.625, "completions/min_length": 154.0, "completions/max_length": 166.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 160.625, "completions/min_terminated_length": 154.0, "completions/max_terminated_length": 166.0, "rewards/meter/mean": 0.9891646504402161, "rewards/meter/std": 0.004805354867130518, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.9540842771530151, "rewards/total_composite/std": 0.06925938278436661, "reward": 0.9540842771530151, "reward_std": 0.06925938278436661, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.037685707211494446, "sampling/sampling_logp_difference/max": 0.9908924102783203, "sampling/importance_sampling_ratio/min": 0.37124526500701904, "sampling/importance_sampling_ratio/mean": 1.0097475051879883, "sampling/importance_sampling_ratio/max": 1.8647836446762085, "entropy": 0.38958779722452164, "clip_ratio/low_mean": 0.007823501946404576, "clip_ratio/low_min": 0.007823501946404576, "clip_ratio/high_mean": 0.02487414190545678, "clip_ratio/high_max": 0.02487414190545678, "clip_ratio/region_mean": 0.03269764385186136, "reward_total_mean": 0.9540842771530151, "reward_meter_mean": 0.9891646504402161, "reward_meter_std": 0.004805354867130518, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.9540842771530151, "reward_total_composite_std": 0.06925938278436661} {"timestamp_utc": "2026-04-12T01:01:50Z", "mode": "train", "global_step": 1884, "epoch": 0.07567176768285336, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.293939393939394e-06, "num_tokens": 4246477.0, "completions/mean_length": 74.75, "completions/min_length": 73.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.75, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9963148832321167, "rewards/meter/std": 0.0056813484989106655, "rewards/count_adherence/mean": 0.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.0, "rewards/total_composite/std": 0.0, "reward": 0.0, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.04152185842394829, "sampling/sampling_logp_difference/max": 1.5260066986083984, "sampling/importance_sampling_ratio/min": 0.21740210056304932, "sampling/importance_sampling_ratio/mean": 1.003008246421814, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2754708882421255, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.0, "reward_meter_mean": 0.9963148832321167, "reward_meter_std": 0.0056813484989106655, "reward_count_adherence_mean": 0.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.0, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:01:57Z", "mode": "train", "global_step": 1885, "epoch": 0.07571193316463831, "loss": -0.0235, "grad_norm": 2.4556612968444824, "learning_rate": 4.290909090909091e-06, "num_tokens": 4250056.0, "completions/mean_length": 247.375, "completions/min_length": 234.0, "completions/max_length": 260.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 247.375, "completions/min_terminated_length": 234.0, "completions/max_terminated_length": 260.0, "rewards/meter/mean": 0.9978925585746765, "rewards/meter/std": 0.00047845288645476103, "rewards/count_adherence/mean": 0.9464285969734192, "rewards/count_adherence/std": 0.07393559068441391, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8693909645080566, "rewards/repeat_penalty/std": 0.11218275874853134, "rewards/total_composite/mean": 0.8244522213935852, "rewards/total_composite/std": 0.14857427775859833, "reward": 0.8244522213935852, "reward_std": 0.14857426285743713, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031580884009599686, "sampling/sampling_logp_difference/max": 1.2896442413330078, "sampling/importance_sampling_ratio/min": 0.27536872029304504, "sampling/importance_sampling_ratio/mean": 1.0049325227737427, "sampling/importance_sampling_ratio/max": 1.8980882167816162, "entropy": 0.29367757216095924, "clip_ratio/low_mean": 0.010505896178074181, "clip_ratio/low_min": 0.010505896178074181, "clip_ratio/high_mean": 0.01472888421267271, "clip_ratio/high_max": 0.01472888421267271, "clip_ratio/region_mean": 0.02523478039074689, "reward_total_mean": 0.8244522213935852, "reward_meter_mean": 0.9978925585746765, "reward_meter_std": 0.00047845288645476103, "reward_count_adherence_mean": 0.9464285969734192, "reward_count_adherence_std": 0.07393559068441391, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8693909645080566, "reward_repeat_penalty_std": 0.11218275874853134, "reward_total_composite_mean": 0.8244522213935852, "reward_total_composite_std": 0.14857427775859833} {"timestamp_utc": "2026-04-12T01:02:03Z", "mode": "train", "global_step": 1886, "epoch": 0.07575209864642327, "loss": -0.0178, "grad_norm": 2.436864137649536, "learning_rate": 4.287878787878788e-06, "num_tokens": 4253377.0, "completions/mean_length": 214.125, "completions/min_length": 204.0, "completions/max_length": 222.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 214.125, "completions/min_terminated_length": 204.0, "completions/max_terminated_length": 222.0, "rewards/meter/mean": 0.9979674220085144, "rewards/meter/std": 0.00041360704926773906, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.07715168595314026, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8818181753158569, "rewards/repeat_penalty/std": 0.07008175551891327, "rewards/total_composite/mean": 0.8467762470245361, "rewards/total_composite/std": 0.12215393781661987, "reward": 0.8467762470245361, "reward_std": 0.12215393036603928, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03404805809259415, "sampling/sampling_logp_difference/max": 2.1384077072143555, "sampling/importance_sampling_ratio/min": 0.11784233152866364, "sampling/importance_sampling_ratio/mean": 1.0003207921981812, "sampling/importance_sampling_ratio/max": 1.72610342502594, "entropy": 0.27855938114225864, "clip_ratio/low_mean": 0.00671045109629631, "clip_ratio/low_min": 0.00671045109629631, "clip_ratio/high_mean": 0.020219704834744334, "clip_ratio/high_max": 0.020219704834744334, "clip_ratio/region_mean": 0.026930155931040645, "reward_total_mean": 0.8467762470245361, "reward_meter_mean": 0.9979674220085144, "reward_meter_std": 0.00041360704926773906, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.07715168595314026, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8818181753158569, "reward_repeat_penalty_std": 0.07008175551891327, "reward_total_composite_mean": 0.8467762470245361, "reward_total_composite_std": 0.12215393781661987} {"timestamp_utc": "2026-04-12T01:02:09Z", "mode": "train", "global_step": 1887, "epoch": 0.07579226412820822, "loss": -0.0062, "grad_norm": 2.4621639251708984, "learning_rate": 4.284848484848485e-06, "num_tokens": 4255950.0, "completions/mean_length": 144.625, "completions/min_length": 141.0, "completions/max_length": 149.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 144.625, "completions/min_terminated_length": 141.0, "completions/max_terminated_length": 149.0, "rewards/meter/mean": 0.997658908367157, "rewards/meter/std": 0.0006960767204873264, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8928571939468384, "rewards/repeat_penalty/std": 0.10101525485515594, "rewards/total_composite/mean": 0.8907864093780518, "rewards/total_composite/std": 0.10098132491111755, "reward": 0.8907864093780518, "reward_std": 0.10098133236169815, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02850562147796154, "sampling/sampling_logp_difference/max": 1.2042560577392578, "sampling/importance_sampling_ratio/min": 0.29991504549980164, "sampling/importance_sampling_ratio/mean": 1.0084306001663208, "sampling/importance_sampling_ratio/max": 1.7592705488204956, "entropy": 0.25286005809903145, "clip_ratio/low_mean": 0.01478908269200474, "clip_ratio/low_min": 0.01478908269200474, "clip_ratio/high_mean": 0.006850266130641103, "clip_ratio/high_max": 0.006850266130641103, "clip_ratio/region_mean": 0.021639348822645843, "reward_total_mean": 0.8907864093780518, "reward_meter_mean": 0.997658908367157, "reward_meter_std": 0.0006960767204873264, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8928571939468384, "reward_repeat_penalty_std": 0.10101525485515594, "reward_total_composite_mean": 0.8907864093780518, "reward_total_composite_std": 0.10098132491111755} {"timestamp_utc": "2026-04-12T01:02:16Z", "mode": "train", "global_step": 1888, "epoch": 0.07583242960999317, "loss": -0.0096, "grad_norm": 2.372718334197998, "learning_rate": 4.281818181818182e-06, "num_tokens": 4259851.0, "completions/mean_length": 296.625, "completions/min_length": 279.0, "completions/max_length": 317.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 296.625, "completions/min_terminated_length": 279.0, "completions/max_terminated_length": 317.0, "rewards/meter/mean": 0.9935998320579529, "rewards/meter/std": 0.005851938389241695, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.0578637570142746, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8760073184967041, "rewards/repeat_penalty/std": 0.07234964519739151, "rewards/total_composite/mean": 0.7886214256286621, "rewards/total_composite/std": 0.07851462066173553, "reward": 0.7886214256286621, "reward_std": 0.07851462066173553, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.034045975655317307, "sampling/sampling_logp_difference/max": 2.5086705684661865, "sampling/importance_sampling_ratio/min": 0.08137635141611099, "sampling/importance_sampling_ratio/mean": 1.0055367946624756, "sampling/importance_sampling_ratio/max": 1.7818008661270142, "entropy": 0.28176482766866684, "clip_ratio/low_mean": 0.005528143374249339, "clip_ratio/low_min": 0.005528143374249339, "clip_ratio/high_mean": 0.019573589437641203, "clip_ratio/high_max": 0.019573589437641203, "clip_ratio/region_mean": 0.025101732811890543, "reward_total_mean": 0.7886214256286621, "reward_meter_mean": 0.9935998320579529, "reward_meter_std": 0.005851938389241695, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.0578637570142746, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8760073184967041, "reward_repeat_penalty_std": 0.07234964519739151, "reward_total_composite_mean": 0.7886214256286621, "reward_total_composite_std": 0.07851462066173553} {"timestamp_utc": "2026-04-12T01:02:24Z", "mode": "train", "global_step": 1889, "epoch": 0.07587259509177813, "loss": -0.0161, "grad_norm": 1.960667610168457, "learning_rate": 4.278787878787879e-06, "num_tokens": 4264323.0, "completions/mean_length": 328.0, "completions/min_length": 305.0, "completions/max_length": 341.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 328.0, "completions/min_terminated_length": 305.0, "completions/max_terminated_length": 341.0, "rewards/meter/mean": 0.9920902252197266, "rewards/meter/std": 0.005891723092645407, "rewards/count_adherence/mean": 0.6057692766189575, "rewards/count_adherence/std": 0.027196412906050682, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9280506372451782, "rewards/repeat_penalty/std": 0.05198076739907265, "rewards/total_composite/mean": 0.5576076507568359, "rewards/total_composite/std": 0.03838847950100899, "reward": 0.5576076507568359, "reward_std": 0.03838847950100899, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03868624567985535, "sampling/sampling_logp_difference/max": 2.517657995223999, "sampling/importance_sampling_ratio/min": 0.0806482657790184, "sampling/importance_sampling_ratio/mean": 1.0100011825561523, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.35988450422883034, "clip_ratio/low_mean": 0.010531234613154083, "clip_ratio/low_min": 0.010531234613154083, "clip_ratio/high_mean": 0.02012083982117474, "clip_ratio/high_max": 0.02012083982117474, "clip_ratio/region_mean": 0.030652074434328824, "reward_total_mean": 0.5576076507568359, "reward_meter_mean": 0.9920902252197266, "reward_meter_std": 0.005891723092645407, "reward_count_adherence_mean": 0.6057692766189575, "reward_count_adherence_std": 0.027196412906050682, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9280506372451782, "reward_repeat_penalty_std": 0.05198076739907265, "reward_total_composite_mean": 0.5576076507568359, "reward_total_composite_std": 0.03838847950100899} {"timestamp_utc": "2026-04-12T01:02:29Z", "mode": "train", "global_step": 1890, "epoch": 0.07591276057356308, "loss": -0.0005, "grad_norm": 1.555624008178711, "learning_rate": 4.275757575757576e-06, "num_tokens": 4266105.0, "completions/mean_length": 62.75, "completions/min_length": 61.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.75, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9969385862350464, "rewards/meter/std": 0.00011288504174444824, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9969385862350464, "rewards/total_composite/std": 0.00011288504174444824, "reward": 0.9969385862350464, "reward_std": 0.0001128727599279955, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005147209390997887, "sampling/sampling_logp_difference/max": 0.49872732162475586, "sampling/importance_sampling_ratio/min": 0.7584096193313599, "sampling/importance_sampling_ratio/mean": 1.002489447593689, "sampling/importance_sampling_ratio/max": 1.6466243267059326, "entropy": 0.02277997531928122, "clip_ratio/low_mean": 0.0020491802133619785, "clip_ratio/low_min": 0.0020491802133619785, "clip_ratio/high_mean": 0.003968254197388887, "clip_ratio/high_max": 0.003968254197388887, "clip_ratio/region_mean": 0.006017434410750866, "reward_total_mean": 0.9969385862350464, "reward_meter_mean": 0.9969385862350464, "reward_meter_std": 0.00011288504174444824, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9969385862350464, "reward_total_composite_std": 0.00011288504174444824} {"timestamp_utc": "2026-04-12T01:02:33Z", "mode": "train", "global_step": 1891, "epoch": 0.07595292605534804, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.272727272727273e-06, "num_tokens": 4267721.0, "completions/mean_length": 58.0, "completions/min_length": 58.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9946010708808899, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9946010708808899, "rewards/total_composite/std": 0.0, "reward": 0.9946010708808899, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0011367781553417444, "sampling/sampling_logp_difference/max": 0.0915985256433487, "sampling/importance_sampling_ratio/min": 0.9592980146408081, "sampling/importance_sampling_ratio/mean": 1.0008796453475952, "sampling/importance_sampling_ratio/max": 1.095924735069275, "entropy": 0.011020259640645236, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9946010708808899, "reward_meter_mean": 0.9946010708808899, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9946010708808899, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:02:37Z", "mode": "train", "global_step": 1892, "epoch": 0.07599309153713299, "loss": 0.0082, "grad_norm": 4.663846969604492, "learning_rate": 4.2696969696969695e-06, "num_tokens": 4269442.0, "completions/mean_length": 33.125, "completions/min_length": 33.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.125, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9981597661972046, "rewards/meter/std": 0.0018737530335783958, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981597661972046, "rewards/total_composite/std": 0.0018737530335783958, "reward": 0.9981597661972046, "reward_std": 0.0018737498903647065, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011638238094747066, "sampling/sampling_logp_difference/max": 0.7902803421020508, "sampling/importance_sampling_ratio/min": 0.4537176191806793, "sampling/importance_sampling_ratio/mean": 0.9993104338645935, "sampling/importance_sampling_ratio/max": 1.9366061687469482, "entropy": 0.02548716589808464, "clip_ratio/low_mean": 0.0036764706019312143, "clip_ratio/low_min": 0.0036764706019312143, "clip_ratio/high_mean": 0.007575757801532745, "clip_ratio/high_max": 0.007575757801532745, "clip_ratio/region_mean": 0.01125222840346396, "reward_total_mean": 0.9981597661972046, "reward_meter_mean": 0.9981597661972046, "reward_meter_std": 0.0018737530335783958, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981597661972046, "reward_total_composite_std": 0.0018737530335783958} {"timestamp_utc": "2026-04-12T01:02:43Z", "mode": "train", "global_step": 1893, "epoch": 0.07603325701891794, "loss": 0.012, "grad_norm": 4.968896865844727, "learning_rate": 4.266666666666668e-06, "num_tokens": 4272121.0, "completions/mean_length": 147.875, "completions/min_length": 142.0, "completions/max_length": 154.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 147.875, "completions/min_terminated_length": 142.0, "completions/max_terminated_length": 154.0, "rewards/meter/mean": 0.997498095035553, "rewards/meter/std": 0.0016895781736820936, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9796580076217651, "rewards/total_composite/std": 0.0498599074780941, "reward": 0.9796580076217651, "reward_std": 0.049859896302223206, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05557761713862419, "sampling/sampling_logp_difference/max": 1.7668085098266602, "sampling/importance_sampling_ratio/min": 0.17087747156620026, "sampling/importance_sampling_ratio/mean": 1.002869725227356, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3808891177177429, "clip_ratio/low_mean": 0.004999999888241291, "clip_ratio/low_min": 0.004999999888241291, "clip_ratio/high_mean": 0.043705128598958254, "clip_ratio/high_max": 0.043705128598958254, "clip_ratio/region_mean": 0.048705128487199545, "reward_total_mean": 0.9796580076217651, "reward_meter_mean": 0.997498095035553, "reward_meter_std": 0.0016895781736820936, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.9796580076217651, "reward_total_composite_std": 0.0498599074780941} {"timestamp_utc": "2026-04-12T01:02:49Z", "mode": "train", "global_step": 1894, "epoch": 0.0760734225007029, "loss": 0.01, "grad_norm": 3.8838560581207275, "learning_rate": 4.263636363636364e-06, "num_tokens": 4274755.0, "completions/mean_length": 184.25, "completions/min_length": 177.0, "completions/max_length": 189.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 184.25, "completions/min_terminated_length": 177.0, "completions/max_terminated_length": 189.0, "rewards/meter/mean": 0.9978748559951782, "rewards/meter/std": 0.0017333579016849399, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9444444179534912, "rewards/repeat_penalty/std": 0.08399210125207901, "rewards/total_composite/mean": 0.9423543214797974, "rewards/total_composite/std": 0.08282524347305298, "reward": 0.9423543214797974, "reward_std": 0.08282525092363358, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.049167633056640625, "sampling/sampling_logp_difference/max": 2.1750478744506836, "sampling/importance_sampling_ratio/min": 0.11360271275043488, "sampling/importance_sampling_ratio/mean": 1.002802848815918, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.33961667120456696, "clip_ratio/low_mean": 0.012207416351884604, "clip_ratio/low_min": 0.012207416351884604, "clip_ratio/high_mean": 0.03263740334659815, "clip_ratio/high_max": 0.03263740334659815, "clip_ratio/region_mean": 0.04484481969848275, "reward_total_mean": 0.9423543214797974, "reward_meter_mean": 0.9978748559951782, "reward_meter_std": 0.0017333579016849399, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9444444179534912, "reward_repeat_penalty_std": 0.08399210125207901, "reward_total_composite_mean": 0.9423543214797974, "reward_total_composite_std": 0.08282524347305298} {"timestamp_utc": "2026-04-12T01:02:55Z", "mode": "train", "global_step": 1895, "epoch": 0.07611358798248785, "loss": -0.0041, "grad_norm": 2.22912859916687, "learning_rate": 4.260606060606061e-06, "num_tokens": 4277761.0, "completions/mean_length": 200.75, "completions/min_length": 195.0, "completions/max_length": 207.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 200.75, "completions/min_terminated_length": 195.0, "completions/max_terminated_length": 207.0, "rewards/meter/mean": 0.9934834837913513, "rewards/meter/std": 0.003708732081577182, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.8416043519973755, "rewards/total_composite/std": 0.34394556283950806, "reward": 0.8416043519973755, "reward_std": 0.34394556283950806, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.037548985332250595, "sampling/sampling_logp_difference/max": 1.3242321014404297, "sampling/importance_sampling_ratio/min": 0.2660071551799774, "sampling/importance_sampling_ratio/mean": 1.0138887166976929, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3611125461757183, "clip_ratio/low_mean": 0.0031887756194919348, "clip_ratio/low_min": 0.0031887756194919348, "clip_ratio/high_mean": 0.022966360847931355, "clip_ratio/high_max": 0.022966360847931355, "clip_ratio/region_mean": 0.02615513646742329, "reward_total_mean": 0.8416043519973755, "reward_meter_mean": 0.9934834837913513, "reward_meter_std": 0.003708732081577182, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_total_composite_mean": 0.8416043519973755, "reward_total_composite_std": 0.34394556283950806} {"timestamp_utc": "2026-04-12T01:03:00Z", "mode": "train", "global_step": 1896, "epoch": 0.0761537534642728, "loss": 0.0005, "grad_norm": 0.0076941619627177715, "learning_rate": 4.2575757575757585e-06, "num_tokens": 4280665.0, "completions/mean_length": 165.0, "completions/min_length": 165.0, "completions/max_length": 165.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 165.0, "completions/min_terminated_length": 165.0, "completions/max_terminated_length": 165.0, "rewards/meter/mean": 0.9991617202758789, "rewards/meter/std": 2.346204610148561e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7777777910232544, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7771258354187012, "rewards/total_composite/std": 1.8245253841087106e-06, "reward": 0.7771258354187012, "reward_std": 1.8390023797110189e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0018403534777462482, "sampling/sampling_logp_difference/max": 1.3649120330810547, "sampling/importance_sampling_ratio/min": 0.25540316104888916, "sampling/importance_sampling_ratio/mean": 1.0000821352005005, "sampling/importance_sampling_ratio/max": 1.0792498588562012, "entropy": 0.007689857040531933, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0007575757335871458, "clip_ratio/high_max": 0.0007575757335871458, "clip_ratio/region_mean": 0.0007575757335871458, "reward_total_mean": 0.7771258354187012, "reward_meter_mean": 0.9991617202758789, "reward_meter_std": 2.346204610148561e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7777777910232544, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7771258354187012, "reward_total_composite_std": 1.8245253841087106e-06} {"timestamp_utc": "2026-04-12T01:03:06Z", "mode": "train", "global_step": 1897, "epoch": 0.07619391894605776, "loss": 0.0006, "grad_norm": 4.121653079986572, "learning_rate": 4.254545454545455e-06, "num_tokens": 4283357.0, "completions/mean_length": 152.5, "completions/min_length": 144.0, "completions/max_length": 158.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 152.5, "completions/min_terminated_length": 144.0, "completions/max_terminated_length": 158.0, "rewards/meter/mean": 0.9909997582435608, "rewards/meter/std": 0.008123436011373997, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9732140302658081, "rewards/total_composite/std": 0.04892229288816452, "reward": 0.9732140302658081, "reward_std": 0.04892229288816452, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04408065602183342, "sampling/sampling_logp_difference/max": 2.633777618408203, "sampling/importance_sampling_ratio/min": 0.07180669158697128, "sampling/importance_sampling_ratio/mean": 1.006994605064392, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.34401146695017815, "clip_ratio/low_mean": 0.0025167784187942743, "clip_ratio/low_min": 0.0025167784187942743, "clip_ratio/high_mean": 0.04165800241753459, "clip_ratio/high_max": 0.04165800241753459, "clip_ratio/region_mean": 0.044174780836328864, "reward_total_mean": 0.9732140302658081, "reward_meter_mean": 0.9909997582435608, "reward_meter_std": 0.008123436011373997, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.9732140302658081, "reward_total_composite_std": 0.04892229288816452} {"timestamp_utc": "2026-04-12T01:03:11Z", "mode": "train", "global_step": 1898, "epoch": 0.07623408442784271, "loss": -0.001, "grad_norm": 2.702057361602783, "learning_rate": 4.251515151515152e-06, "num_tokens": 4285658.0, "completions/mean_length": 119.625, "completions/min_length": 116.0, "completions/max_length": 123.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 119.625, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.9938209652900696, "rewards/meter/std": 0.002714785048738122, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9938209652900696, "rewards/total_composite/std": 0.002714785048738122, "reward": 0.9938209652900696, "reward_std": 0.0027147859800606966, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.039888497442007065, "sampling/sampling_logp_difference/max": 1.169196605682373, "sampling/importance_sampling_ratio/min": 0.3106164038181305, "sampling/importance_sampling_ratio/mean": 1.0123997926712036, "sampling/importance_sampling_ratio/max": 1.5793664455413818, "entropy": 0.3594598062336445, "clip_ratio/low_mean": 0.017732840729877353, "clip_ratio/low_min": 0.017732840729877353, "clip_ratio/high_mean": 0.01466671982780099, "clip_ratio/high_max": 0.01466671982780099, "clip_ratio/region_mean": 0.03239956055767834, "reward_total_mean": 0.9938209652900696, "reward_meter_mean": 0.9938209652900696, "reward_meter_std": 0.002714785048738122, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9938209652900696, "reward_total_composite_std": 0.002714785048738122} {"timestamp_utc": "2026-04-12T01:03:16Z", "mode": "train", "global_step": 1899, "epoch": 0.07627424990962767, "loss": 0.0146, "grad_norm": 5.739034175872803, "learning_rate": 4.248484848484849e-06, "num_tokens": 4287482.0, "completions/mean_length": 63.0, "completions/min_length": 62.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9764682650566101, "rewards/meter/std": 0.05796706676483154, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9764682650566101, "rewards/total_composite/std": 0.05796706676483154, "reward": 0.9764682650566101, "reward_std": 0.057967059314250946, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010971495881676674, "sampling/sampling_logp_difference/max": 1.089432716369629, "sampling/importance_sampling_ratio/min": 0.33640727400779724, "sampling/importance_sampling_ratio/mean": 0.9993007183074951, "sampling/importance_sampling_ratio/max": 1.515744924545288, "entropy": 0.05075907311402261, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.002016128972172737, "clip_ratio/high_max": 0.002016128972172737, "clip_ratio/region_mean": 0.002016128972172737, "reward_total_mean": 0.9764682650566101, "reward_meter_mean": 0.9764682650566101, "reward_meter_std": 0.05796706676483154, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9764682650566101, "reward_total_composite_std": 0.05796706676483154} {"timestamp_utc": "2026-04-12T01:03:21Z", "mode": "train", "global_step": 1900, "epoch": 0.07631441539141262, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.245454545454546e-06, "num_tokens": 4289610.0, "completions/mean_length": 99.0, "completions/min_length": 99.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.0, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.9990620613098145, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990620613098145, "rewards/total_composite/std": 0.0, "reward": 0.9990620613098145, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0004978616489097476, "sampling/sampling_logp_difference/max": 0.014289772137999535, "sampling/importance_sampling_ratio/min": 0.9927101135253906, "sampling/importance_sampling_ratio/mean": 1.00044846534729, "sampling/importance_sampling_ratio/max": 1.0143922567367554, "entropy": 0.004198343900498003, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9990620613098145, "reward_meter_mean": 0.9990620613098145, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990620613098145, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:04:25Z", "mode": "eval", "global_step": 1900, "epoch": 0.07631441539141262, "eval_loss": NaN, "eval_runtime": 64.1724, "eval_samples_per_second": 1.621, "eval_steps_per_second": 0.203, "eval_num_tokens": 4289610.0, "eval_completions/mean_length": 199.51923076923077, "eval_completions/min_length": 71.53846153846153, "eval_completions/max_length": 335.6923076923077, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 199.51923076923077, "eval_completions/min_terminated_length": 71.53846153846153, "eval_completions/max_terminated_length": 335.6923076923077, "eval_rewards/meter/mean": 0.7173460401021517, "eval_rewards/meter/std": 0.4227825838785905, "eval_rewards/count_adherence/mean": 0.8646397361388574, "eval_rewards/count_adherence/std": 0.19298870517657354, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/repeat_penalty/mean": 0.8317501590802119, "eval_rewards/repeat_penalty/std": 0.14178751237117326, "eval_rewards/total_composite/mean": 0.5190799373846787, "eval_rewards/total_composite/std": 0.36580440631279576, "eval_reward": 0.5190799373846787, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.014502167307700101, "eval_sampling/sampling_logp_difference/max": 0.989627324617826, "eval_sampling/importance_sampling_ratio/min": 0.3849313259124756, "eval_sampling/importance_sampling_ratio/mean": 1.0033589509817271, "eval_sampling/importance_sampling_ratio/max": 1.388241483614995, "eval_entropy": 0.15330308188612646, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5190799373846787, "eval_reward_meter_mean": 0.7173460401021517, "eval_reward_meter_std": 0.4227825838785905, "eval_reward_count_adherence_mean": 0.8646397361388574, "eval_reward_count_adherence_std": 0.19298870517657354, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_repeat_penalty_mean": 0.8317501590802119, "eval_reward_repeat_penalty_std": 0.14178751237117326, "eval_reward_total_composite_mean": 0.5190799373846787, "eval_reward_total_composite_std": 0.36580440631279576} {"timestamp_utc": "2026-04-12T01:04:32Z", "mode": "train", "global_step": 1901, "epoch": 0.07635458087319758, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.242424242424243e-06, "num_tokens": 4291354.0, "completions/mean_length": 58.0, "completions/min_length": 58.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9946010708808899, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9946010708808899, "rewards/total_composite/std": 0.0, "reward": 0.9946010708808899, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0012767898151651025, "sampling/sampling_logp_difference/max": 0.17115801572799683, "sampling/importance_sampling_ratio/min": 0.8426884412765503, "sampling/importance_sampling_ratio/mean": 1.00002121925354, "sampling/importance_sampling_ratio/max": 1.0319656133651733, "entropy": 0.013817269122228026, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9946010708808899, "reward_meter_mean": 0.9946010708808899, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9946010708808899, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:04:36Z", "mode": "train", "global_step": 1902, "epoch": 0.07639474635498253, "loss": 0.0286, "grad_norm": 5.961615085601807, "learning_rate": 4.2393939393939395e-06, "num_tokens": 4293032.0, "completions/mean_length": 50.75, "completions/min_length": 49.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.75, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.9352545142173767, "rewards/meter/std": 0.01202213205397129, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9352545142173767, "rewards/total_composite/std": 0.01202213205397129, "reward": 0.9352545142173767, "reward_std": 0.012022126466035843, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016524022445082664, "sampling/sampling_logp_difference/max": 0.9960556030273438, "sampling/importance_sampling_ratio/min": 0.3693333864212036, "sampling/importance_sampling_ratio/mean": 0.9972783327102661, "sampling/importance_sampling_ratio/max": 1.7655236721038818, "entropy": 0.0905577321536839, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.005102040711790323, "clip_ratio/high_max": 0.005102040711790323, "clip_ratio/region_mean": 0.005102040711790323, "reward_total_mean": 0.9352545142173767, "reward_meter_mean": 0.9352545142173767, "reward_meter_std": 0.01202213205397129, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9352545142173767, "reward_total_composite_std": 0.01202213205397129} {"timestamp_utc": "2026-04-12T01:04:42Z", "mode": "train", "global_step": 1903, "epoch": 0.07643491183676748, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.236363636363637e-06, "num_tokens": 4295672.0, "completions/mean_length": 127.0, "completions/min_length": 127.0, "completions/max_length": 127.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.0, "completions/min_terminated_length": 127.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.9974547624588013, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8549612164497375, "rewards/total_composite/std": 0.0, "reward": 0.8549612164497375, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.000930364360101521, "sampling/sampling_logp_difference/max": 0.02307342365384102, "sampling/importance_sampling_ratio/min": 0.9878833293914795, "sampling/importance_sampling_ratio/mean": 1.0008466243743896, "sampling/importance_sampling_ratio/max": 1.0233416557312012, "entropy": 0.008461171586532146, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8549612164497375, "reward_meter_mean": 0.9974547624588013, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8549612164497375, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:04:46Z", "mode": "train", "global_step": 1904, "epoch": 0.07647507731855244, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.233333333333334e-06, "num_tokens": 4297480.0, "completions/mean_length": 69.0, "completions/min_length": 69.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.0, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.997582197189331, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.997582197189331, "rewards/total_composite/std": 0.0, "reward": 0.997582197189331, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.000656367396004498, "sampling/sampling_logp_difference/max": 0.00906725786626339, "sampling/importance_sampling_ratio/min": 0.990973711013794, "sampling/importance_sampling_ratio/mean": 1.0005828142166138, "sampling/importance_sampling_ratio/max": 1.0086549520492554, "entropy": 0.006679805053863674, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.997582197189331, "reward_meter_mean": 0.997582197189331, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.997582197189331, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:04:51Z", "mode": "train", "global_step": 1905, "epoch": 0.07651524280033739, "loss": 0.0031, "grad_norm": 2.374253034591675, "learning_rate": 4.2303030303030304e-06, "num_tokens": 4299561.0, "completions/mean_length": 95.125, "completions/min_length": 95.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.125, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.9972648620605469, "rewards/meter/std": 0.00012022549344692379, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9972648620605469, "rewards/total_composite/std": 0.00012022549344692379, "reward": 0.9972648620605469, "reward_std": 0.00012023184535792097, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008180440403521061, "sampling/sampling_logp_difference/max": 1.0717182159423828, "sampling/importance_sampling_ratio/min": 0.34241965413093567, "sampling/importance_sampling_ratio/mean": 1.0006614923477173, "sampling/importance_sampling_ratio/max": 1.5949854850769043, "entropy": 0.03163000999484211, "clip_ratio/low_mean": 0.0026041667442768812, "clip_ratio/low_min": 0.0026041667442768812, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0026041667442768812, "reward_total_mean": 0.9972648620605469, "reward_meter_mean": 0.9972648620605469, "reward_meter_std": 0.00012022549344692379, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9972648620605469, "reward_total_composite_std": 0.00012022549344692379} {"timestamp_utc": "2026-04-12T01:04:56Z", "mode": "train", "global_step": 1906, "epoch": 0.07655540828212234, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.227272727272728e-06, "num_tokens": 4301393.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.998939037322998, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.0, "rewards/total_composite/std": 0.0, "reward": 0.0, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0003236948396079242, "sampling/sampling_logp_difference/max": 0.011038463562726974, "sampling/importance_sampling_ratio/min": 0.9955568313598633, "sampling/importance_sampling_ratio/mean": 1.0002886056900024, "sampling/importance_sampling_ratio/max": 1.0110995769500732, "entropy": 0.0030257631733547896, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.0, "reward_meter_mean": 0.998939037322998, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.0, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:05:03Z", "mode": "train", "global_step": 1907, "epoch": 0.0765955737639073, "loss": 0.0114, "grad_norm": 2.0788662433624268, "learning_rate": 4.224242424242425e-06, "num_tokens": 4305238.0, "completions/mean_length": 251.625, "completions/min_length": 243.0, "completions/max_length": 258.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 251.625, "completions/min_terminated_length": 243.0, "completions/max_terminated_length": 258.0, "rewards/meter/mean": 0.997933030128479, "rewards/meter/std": 0.00043991892016492784, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7692307829856873, "rewards/repeat_penalty/std": 0.07121692597866058, "rewards/total_composite/mean": 0.7676311731338501, "rewards/total_composite/std": 0.07094842940568924, "reward": 0.7676311731338501, "reward_std": 0.07094840705394745, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02055802382528782, "sampling/sampling_logp_difference/max": 0.9864959716796875, "sampling/importance_sampling_ratio/min": 0.3728809952735901, "sampling/importance_sampling_ratio/mean": 1.0044364929199219, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17152301780879498, "clip_ratio/low_mean": 0.0019704491132870317, "clip_ratio/low_min": 0.0019704491132870317, "clip_ratio/high_mean": 0.014517532312311232, "clip_ratio/high_max": 0.014517532312311232, "clip_ratio/region_mean": 0.016487981425598264, "reward_total_mean": 0.7676311731338501, "reward_meter_mean": 0.997933030128479, "reward_meter_std": 0.00043991892016492784, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7692307829856873, "reward_repeat_penalty_std": 0.07121692597866058, "reward_total_composite_mean": 0.7676311731338501, "reward_total_composite_std": 0.07094842940568924} {"timestamp_utc": "2026-04-12T01:05:07Z", "mode": "train", "global_step": 1908, "epoch": 0.07663573924569225, "loss": 0.0026, "grad_norm": 4.272599697113037, "learning_rate": 4.221212121212121e-06, "num_tokens": 4307205.0, "completions/mean_length": 73.875, "completions/min_length": 72.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.875, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9967119693756104, "rewards/meter/std": 0.001478171325288713, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9967119693756104, "rewards/total_composite/std": 0.001478171325288713, "reward": 0.9967119693756104, "reward_std": 0.0014781749341636896, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.034489989280700684, "sampling/sampling_logp_difference/max": 1.2823877334594727, "sampling/importance_sampling_ratio/min": 0.2773742079734802, "sampling/importance_sampling_ratio/mean": 1.0084983110427856, "sampling/importance_sampling_ratio/max": 1.7452210187911987, "entropy": 0.2469378486275673, "clip_ratio/low_mean": 0.01705764839425683, "clip_ratio/low_min": 0.01705764839425683, "clip_ratio/high_mean": 0.013561643892899156, "clip_ratio/high_max": 0.013561643892899156, "clip_ratio/region_mean": 0.030619292287155986, "reward_total_mean": 0.9967119693756104, "reward_meter_mean": 0.9967119693756104, "reward_meter_std": 0.001478171325288713, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9967119693756104, "reward_total_composite_std": 0.001478171325288713} {"timestamp_utc": "2026-04-12T01:05:12Z", "mode": "train", "global_step": 1909, "epoch": 0.0766759047274772, "loss": -0.0005, "grad_norm": 0.519289493560791, "learning_rate": 4.218181818181819e-06, "num_tokens": 4309357.0, "completions/mean_length": 103.0, "completions/min_length": 103.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.0, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.9975078105926514, "rewards/meter/std": 3.284694321337156e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9975078105926514, "rewards/total_composite/std": 3.284694321337156e-05, "reward": 0.9975078105926514, "reward_std": 3.283110709162429e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0023622734006494284, "sampling/sampling_logp_difference/max": 0.6333708763122559, "sampling/importance_sampling_ratio/min": 0.5307995080947876, "sampling/importance_sampling_ratio/mean": 1.0008713006973267, "sampling/importance_sampling_ratio/max": 1.714148759841919, "entropy": 0.009892041911371052, "clip_ratio/low_mean": 0.0012135922443121672, "clip_ratio/low_min": 0.0012135922443121672, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0012135922443121672, "reward_total_mean": 0.9975078105926514, "reward_meter_mean": 0.9975078105926514, "reward_meter_std": 3.284694321337156e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9975078105926514, "reward_total_composite_std": 3.284694321337156e-05} {"timestamp_utc": "2026-04-12T01:05:20Z", "mode": "train", "global_step": 1910, "epoch": 0.07671607020926216, "loss": -0.0326, "grad_norm": 1.9287523031234741, "learning_rate": 4.215151515151515e-06, "num_tokens": 4313350.0, "completions/mean_length": 291.125, "completions/min_length": 258.0, "completions/max_length": 307.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 291.125, "completions/min_terminated_length": 258.0, "completions/max_terminated_length": 307.0, "rewards/meter/mean": 0.9932549595832825, "rewards/meter/std": 0.0015650950372219086, "rewards/count_adherence/mean": 0.8888888955116272, "rewards/count_adherence/std": 0.059391383081674576, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6931459903717041, "rewards/repeat_penalty/std": 0.010692903771996498, "rewards/total_composite/mean": 0.6118792295455933, "rewards/total_composite/std": 0.041350506246089935, "reward": 0.6118792295455933, "reward_std": 0.04135050252079964, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010057836771011353, "sampling/sampling_logp_difference/max": 1.9771476984024048, "sampling/importance_sampling_ratio/min": 0.1384636014699936, "sampling/importance_sampling_ratio/mean": 0.9992997646331787, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0457235153298825, "clip_ratio/low_mean": 0.006014678278006613, "clip_ratio/low_min": 0.006014678278006613, "clip_ratio/high_mean": 0.0004071661096531898, "clip_ratio/high_max": 0.0004071661096531898, "clip_ratio/region_mean": 0.006421844387659803, "reward_total_mean": 0.6118792295455933, "reward_meter_mean": 0.9932549595832825, "reward_meter_std": 0.0015650950372219086, "reward_count_adherence_mean": 0.8888888955116272, "reward_count_adherence_std": 0.059391383081674576, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6931459903717041, "reward_repeat_penalty_std": 0.010692903771996498, "reward_total_composite_mean": 0.6118792295455933, "reward_total_composite_std": 0.041350506246089935} {"timestamp_utc": "2026-04-12T01:05:24Z", "mode": "train", "global_step": 1911, "epoch": 0.07675623569104711, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.212121212121212e-06, "num_tokens": 4315262.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.998939037322998, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998939037322998, "rewards/total_composite/std": 0.0, "reward": 0.998939037322998, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0003693441394716501, "sampling/sampling_logp_difference/max": 0.006262333132326603, "sampling/importance_sampling_ratio/min": 0.9994460940361023, "sampling/importance_sampling_ratio/mean": 1.0003656148910522, "sampling/importance_sampling_ratio/max": 1.006282091140747, "entropy": 0.004040350759169087, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.998939037322998, "reward_meter_mean": 0.998939037322998, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.998939037322998, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:05:29Z", "mode": "train", "global_step": 1912, "epoch": 0.07679640117283207, "loss": 0.0001, "grad_norm": 0.2683267891407013, "learning_rate": 4.2090909090909095e-06, "num_tokens": 4317134.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9989374279975891, "rewards/meter/std": 4.551860001811292e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989374279975891, "rewards/total_composite/std": 4.551860001811292e-06, "reward": 0.9989374279975891, "reward_std": 4.551859547063941e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.002090727211907506, "sampling/sampling_logp_difference/max": 0.19945931434631348, "sampling/importance_sampling_ratio/min": 0.8604245185852051, "sampling/importance_sampling_ratio/mean": 1.0016252994537354, "sampling/importance_sampling_ratio/max": 1.2207425832748413, "entropy": 0.011784833506681025, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0018939394503831863, "clip_ratio/high_max": 0.0018939394503831863, "clip_ratio/region_mean": 0.0018939394503831863, "reward_total_mean": 0.9989374279975891, "reward_meter_mean": 0.9989374279975891, "reward_meter_std": 4.551860001811292e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9989374279975891, "reward_total_composite_std": 4.551860001811292e-06} {"timestamp_utc": "2026-04-12T01:05:33Z", "mode": "train", "global_step": 1913, "epoch": 0.07683656665461702, "loss": -0.0002, "grad_norm": 0.7913539409637451, "learning_rate": 4.206060606060606e-06, "num_tokens": 4318894.0, "completions/mean_length": 58.0, "completions/min_length": 58.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9945913553237915, "rewards/meter/std": 2.7374408091418445e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9945913553237915, "rewards/total_composite/std": 2.7374408091418445e-05, "reward": 0.9945913553237915, "reward_std": 2.7371370379114524e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003088480094447732, "sampling/sampling_logp_difference/max": 0.22362565994262695, "sampling/importance_sampling_ratio/min": 0.8597717881202698, "sampling/importance_sampling_ratio/mean": 1.0011041164398193, "sampling/importance_sampling_ratio/max": 1.2506028413772583, "entropy": 0.018943659961223602, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9945913553237915, "reward_meter_mean": 0.9945913553237915, "reward_meter_std": 2.7374408091418445e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9945913553237915, "reward_total_composite_std": 2.7374408091418445e-05} {"timestamp_utc": "2026-04-12T01:05:38Z", "mode": "train", "global_step": 1914, "epoch": 0.07687673213640198, "loss": -0.0005, "grad_norm": 0.36171841621398926, "learning_rate": 4.203030303030303e-06, "num_tokens": 4320646.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9989273548126221, "rewards/meter/std": 3.3127438655355945e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989273548126221, "rewards/total_composite/std": 3.3127438655355945e-05, "reward": 0.9989273548126221, "reward_std": 3.312742410344072e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004210924729704857, "sampling/sampling_logp_difference/max": 1.9241247177124023, "sampling/importance_sampling_ratio/min": 0.14600348472595215, "sampling/importance_sampling_ratio/mean": 0.9987910985946655, "sampling/importance_sampling_ratio/max": 1.0103963613510132, "entropy": 0.005322357552358881, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9989273548126221, "reward_meter_mean": 0.9989273548126221, "reward_meter_std": 3.3127438655355945e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9989273548126221, "reward_total_composite_std": 3.3127438655355945e-05} {"timestamp_utc": "2026-04-12T01:05:43Z", "mode": "train", "global_step": 1915, "epoch": 0.07691689761818693, "loss": -0.0005, "grad_norm": 0.24196073412895203, "learning_rate": 4.2000000000000004e-06, "num_tokens": 4323030.0, "completions/mean_length": 122.0, "completions/min_length": 122.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.0, "completions/min_terminated_length": 122.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.995856523513794, "rewards/meter/std": 2.842021240212489e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8535913228988647, "rewards/total_composite/std": 2.4363118427572772e-05, "reward": 0.8535913228988647, "reward_std": 2.4356924768653698e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00429805601015687, "sampling/sampling_logp_difference/max": 0.30840301513671875, "sampling/importance_sampling_ratio/min": 0.7478480339050293, "sampling/importance_sampling_ratio/mean": 1.0021631717681885, "sampling/importance_sampling_ratio/max": 1.361249566078186, "entropy": 0.02201851806603372, "clip_ratio/low_mean": 0.0010245901066809893, "clip_ratio/low_min": 0.0010245901066809893, "clip_ratio/high_mean": 0.0020491802133619785, "clip_ratio/high_max": 0.0020491802133619785, "clip_ratio/region_mean": 0.003073770320042968, "reward_total_mean": 0.8535913228988647, "reward_meter_mean": 0.995856523513794, "reward_meter_std": 2.842021240212489e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8535913228988647, "reward_total_composite_std": 2.4363118427572772e-05} {"timestamp_utc": "2026-04-12T01:05:48Z", "mode": "train", "global_step": 1916, "epoch": 0.07695706309997188, "loss": 0.0035, "grad_norm": 2.755591630935669, "learning_rate": 4.196969696969697e-06, "num_tokens": 4324864.0, "completions/mean_length": 74.25, "completions/min_length": 72.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.25, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9977705478668213, "rewards/meter/std": 0.0007801271858625114, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977705478668213, "rewards/total_composite/std": 0.0007801271858625114, "reward": 0.9977705478668213, "reward_std": 0.0007801253814250231, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02434798702597618, "sampling/sampling_logp_difference/max": 1.0248136520385742, "sampling/importance_sampling_ratio/min": 0.35886335372924805, "sampling/importance_sampling_ratio/mean": 1.0088934898376465, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.18484536185860634, "clip_ratio/low_mean": 0.00856418942566961, "clip_ratio/low_min": 0.00856418942566961, "clip_ratio/high_mean": 0.005000000121071935, "clip_ratio/high_max": 0.005000000121071935, "clip_ratio/region_mean": 0.013564189546741545, "reward_total_mean": 0.9977705478668213, "reward_meter_mean": 0.9977705478668213, "reward_meter_std": 0.0007801271858625114, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9977705478668213, "reward_total_composite_std": 0.0007801271858625114} {"timestamp_utc": "2026-04-12T01:05:53Z", "mode": "train", "global_step": 1917, "epoch": 0.07699722858175684, "loss": 0.1368, "grad_norm": 7.223612308502197, "learning_rate": 4.193939393939394e-06, "num_tokens": 4326780.0, "completions/mean_length": 81.5, "completions/min_length": 75.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.5, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.9922255277633667, "rewards/meter/std": 0.006029736250638962, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9299638271331787, "rewards/total_composite/std": 0.17459870874881744, "reward": 0.9299638271331787, "reward_std": 0.17459867894649506, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.043613094836473465, "sampling/sampling_logp_difference/max": 3.7613840103149414, "sampling/importance_sampling_ratio/min": 0.02325153723359108, "sampling/importance_sampling_ratio/mean": 1.0046718120574951, "sampling/importance_sampling_ratio/max": 1.5985634326934814, "entropy": 0.30126412957906723, "clip_ratio/low_mean": 0.006756756920367479, "clip_ratio/low_min": 0.006756756920367479, "clip_ratio/high_mean": 0.038703347789123654, "clip_ratio/high_max": 0.038703347789123654, "clip_ratio/region_mean": 0.045460104709491134, "reward_total_mean": 0.9299638271331787, "reward_meter_mean": 0.9922255277633667, "reward_meter_std": 0.006029736250638962, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9299638271331787, "reward_total_composite_std": 0.17459870874881744} {"timestamp_utc": "2026-04-12T01:05:57Z", "mode": "train", "global_step": 1918, "epoch": 0.07703739406354179, "loss": 0.0163, "grad_norm": 3.7035961151123047, "learning_rate": 4.190909090909091e-06, "num_tokens": 4328592.0, "completions/mean_length": 72.5, "completions/min_length": 70.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.5, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.986924946308136, "rewards/meter/std": 0.002686952007934451, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.986924946308136, "rewards/total_composite/std": 0.002686952007934451, "reward": 0.986924946308136, "reward_std": 0.0026869599241763353, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016851795837283134, "sampling/sampling_logp_difference/max": 0.770622730255127, "sampling/importance_sampling_ratio/min": 0.46272483468055725, "sampling/importance_sampling_ratio/mean": 1.0001866817474365, "sampling/importance_sampling_ratio/max": 1.7578001022338867, "entropy": 0.15394389163702726, "clip_ratio/low_mean": 0.005113846738822758, "clip_ratio/low_min": 0.005113846738822758, "clip_ratio/high_mean": 0.017437846050597727, "clip_ratio/high_max": 0.017437846050597727, "clip_ratio/region_mean": 0.022551692789420485, "reward_total_mean": 0.986924946308136, "reward_meter_mean": 0.986924946308136, "reward_meter_std": 0.002686952007934451, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.986924946308136, "reward_total_composite_std": 0.002686952007934451} {"timestamp_utc": "2026-04-12T01:06:02Z", "mode": "train", "global_step": 1919, "epoch": 0.07707755954532675, "loss": -0.0004, "grad_norm": 1.28860342502594, "learning_rate": 4.187878787878788e-06, "num_tokens": 4330512.0, "completions/mean_length": 69.0, "completions/min_length": 69.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.0, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9976222515106201, "rewards/meter/std": 4.285174509277567e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9976222515106201, "rewards/total_composite/std": 4.285174509277567e-05, "reward": 0.9976222515106201, "reward_std": 4.2851730540860444e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006954767741262913, "sampling/sampling_logp_difference/max": 0.8643484711647034, "sampling/importance_sampling_ratio/min": 0.4213259816169739, "sampling/importance_sampling_ratio/mean": 0.9985252022743225, "sampling/importance_sampling_ratio/max": 1.3529547452926636, "entropy": 0.027878023218363523, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.005434782709926367, "clip_ratio/high_max": 0.005434782709926367, "clip_ratio/region_mean": 0.005434782709926367, "reward_total_mean": 0.9976222515106201, "reward_meter_mean": 0.9976222515106201, "reward_meter_std": 4.285174509277567e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9976222515106201, "reward_total_composite_std": 4.285174509277567e-05} {"timestamp_utc": "2026-04-12T01:06:07Z", "mode": "train", "global_step": 1920, "epoch": 0.0771177250271117, "loss": -0.0001, "grad_norm": 0.23397041857242584, "learning_rate": 4.184848484848485e-06, "num_tokens": 4332959.0, "completions/mean_length": 131.875, "completions/min_length": 131.0, "completions/max_length": 132.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.875, "completions/min_terminated_length": 131.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.9991158246994019, "rewards/meter/std": 1.4611580809287261e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8563849925994873, "rewards/total_composite/std": 1.2507220162660815e-05, "reward": 0.8563849925994873, "reward_std": 1.2510253327491228e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004140977747738361, "sampling/sampling_logp_difference/max": 1.6029167175292969, "sampling/importance_sampling_ratio/min": 0.2013085037469864, "sampling/importance_sampling_ratio/mean": 0.9987531304359436, "sampling/importance_sampling_ratio/max": 1.0698660612106323, "entropy": 0.01177009951788932, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8563849925994873, "reward_meter_mean": 0.9991158246994019, "reward_meter_std": 1.4611580809287261e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8563849925994873, "reward_total_composite_std": 1.2507220162660815e-05} {"timestamp_utc": "2026-04-12T01:06:12Z", "mode": "train", "global_step": 1921, "epoch": 0.07715789050889665, "loss": 0.0041, "grad_norm": 3.0183725357055664, "learning_rate": 4.181818181818182e-06, "num_tokens": 4335289.0, "completions/mean_length": 114.25, "completions/min_length": 113.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.25, "completions/min_terminated_length": 113.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.9916590452194214, "rewards/meter/std": 0.004383288789540529, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9916590452194214, "rewards/total_composite/std": 0.004383288789540529, "reward": 0.9916590452194214, "reward_std": 0.004383287392556667, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03317651152610779, "sampling/sampling_logp_difference/max": 0.9611685276031494, "sampling/importance_sampling_ratio/min": 0.4941423237323761, "sampling/importance_sampling_ratio/mean": 1.0076040029525757, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3176443073898554, "clip_ratio/low_mean": 0.011023133061826229, "clip_ratio/low_min": 0.011023133061826229, "clip_ratio/high_mean": 0.02495911391451955, "clip_ratio/high_max": 0.02495911391451955, "clip_ratio/region_mean": 0.03598224697634578, "reward_total_mean": 0.9916590452194214, "reward_meter_mean": 0.9916590452194214, "reward_meter_std": 0.004383288789540529, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9916590452194214, "reward_total_composite_std": 0.004383288789540529} {"timestamp_utc": "2026-04-12T01:06:17Z", "mode": "train", "global_step": 1922, "epoch": 0.07719805599068161, "loss": 0.0006, "grad_norm": 0.0710740014910698, "learning_rate": 4.1787878787878795e-06, "num_tokens": 4337264.0, "completions/mean_length": 65.875, "completions/min_length": 65.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.875, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9989427924156189, "rewards/meter/std": 1.0663165085134096e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989427924156189, "rewards/total_composite/std": 1.0663165085134096e-05, "reward": 0.9989427924156189, "reward_std": 1.0657141501724254e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006586765870451927, "sampling/sampling_logp_difference/max": 2.554171562194824, "sampling/importance_sampling_ratio/min": 0.0777566209435463, "sampling/importance_sampling_ratio/mean": 0.9995160102844238, "sampling/importance_sampling_ratio/max": 1.1512835025787354, "entropy": 0.015987101825885475, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.001923076924867928, "clip_ratio/high_max": 0.001923076924867928, "clip_ratio/region_mean": 0.001923076924867928, "reward_total_mean": 0.9989427924156189, "reward_meter_mean": 0.9989427924156189, "reward_meter_std": 1.0663165085134096e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9989427924156189, "reward_total_composite_std": 1.0663165085134096e-05} {"timestamp_utc": "2026-04-12T01:06:23Z", "mode": "train", "global_step": 1923, "epoch": 0.07723822147246656, "loss": 0.0104, "grad_norm": 0.6413597464561462, "learning_rate": 4.175757575757576e-06, "num_tokens": 4340257.0, "completions/mean_length": 201.125, "completions/min_length": 195.0, "completions/max_length": 202.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 201.125, "completions/min_terminated_length": 195.0, "completions/max_terminated_length": 202.0, "rewards/meter/mean": 0.9961963891983032, "rewards/meter/std": 7.053603621898219e-05, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7291666865348816, "rewards/repeat_penalty/std": 0.058925554156303406, "rewards/total_composite/mean": 0.5811127424240112, "rewards/total_composite/std": 0.0469353049993515, "reward": 0.5811127424240112, "reward_std": 0.0469353049993515, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00489839119836688, "sampling/sampling_logp_difference/max": 0.9933547973632812, "sampling/importance_sampling_ratio/min": 0.3703322112560272, "sampling/importance_sampling_ratio/mean": 0.9994967579841614, "sampling/importance_sampling_ratio/max": 1.3452818393707275, "entropy": 0.01974188955500722, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0031607006676495075, "clip_ratio/high_max": 0.0031607006676495075, "clip_ratio/region_mean": 0.0031607006676495075, "reward_total_mean": 0.5811127424240112, "reward_meter_mean": 0.9961963891983032, "reward_meter_std": 7.053603621898219e-05, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7291666865348816, "reward_repeat_penalty_std": 0.058925554156303406, "reward_total_composite_mean": 0.5811127424240112, "reward_total_composite_std": 0.0469353049993515} {"timestamp_utc": "2026-04-12T01:06:28Z", "mode": "train", "global_step": 1924, "epoch": 0.07727838695425152, "loss": 0.1384, "grad_norm": 6.834925174713135, "learning_rate": 4.172727272727273e-06, "num_tokens": 4342199.0, "completions/mean_length": 80.75, "completions/min_length": 72.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.75, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.9895592927932739, "rewards/meter/std": 0.01372271403670311, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.928816556930542, "rewards/total_composite/std": 0.17933166027069092, "reward": 0.928816556930542, "reward_std": 0.17933164536952972, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.048581015318632126, "sampling/sampling_logp_difference/max": 4.14417028427124, "sampling/importance_sampling_ratio/min": 0.015856586396694183, "sampling/importance_sampling_ratio/mean": 1.009357213973999, "sampling/importance_sampling_ratio/max": 1.9494916200637817, "entropy": 0.3375435955822468, "clip_ratio/low_mean": 0.004587155766785145, "clip_ratio/low_min": 0.004587155766785145, "clip_ratio/high_mean": 0.040673923096619546, "clip_ratio/high_max": 0.040673923096619546, "clip_ratio/region_mean": 0.04526107886340469, "reward_total_mean": 0.928816556930542, "reward_meter_mean": 0.9895592927932739, "reward_meter_std": 0.01372271403670311, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.928816556930542, "reward_total_composite_std": 0.17933166027069092} {"timestamp_utc": "2026-04-12T01:06:32Z", "mode": "train", "global_step": 1925, "epoch": 0.07731855243603647, "loss": 0.0103, "grad_norm": 16.744361877441406, "learning_rate": 4.1696969696969705e-06, "num_tokens": 4343965.0, "completions/mean_length": 57.75, "completions/min_length": 56.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.75, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9946485757827759, "rewards/meter/std": 0.00013940427743364125, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9946485757827759, "rewards/total_composite/std": 0.00013940427743364125, "reward": 0.9946485757827759, "reward_std": 0.0001394047576468438, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004949766676872969, "sampling/sampling_logp_difference/max": 0.48597773909568787, "sampling/importance_sampling_ratio/min": 0.6150954961776733, "sampling/importance_sampling_ratio/mean": 0.9992939233779907, "sampling/importance_sampling_ratio/max": 1.1374722719192505, "entropy": 0.025414136587642133, "clip_ratio/low_mean": 0.004387315362691879, "clip_ratio/low_min": 0.004387315362691879, "clip_ratio/high_mean": 0.004310344811528921, "clip_ratio/high_max": 0.004310344811528921, "clip_ratio/region_mean": 0.0086976601742208, "reward_total_mean": 0.9946485757827759, "reward_meter_mean": 0.9946485757827759, "reward_meter_std": 0.00013940427743364125, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9946485757827759, "reward_total_composite_std": 0.00013940427743364125} {"timestamp_utc": "2026-04-12T01:06:40Z", "mode": "train", "global_step": 1926, "epoch": 0.07735871791782142, "loss": 0.0014, "grad_norm": 2.593313694000244, "learning_rate": 4.166666666666667e-06, "num_tokens": 4347997.0, "completions/mean_length": 297.0, "completions/min_length": 282.0, "completions/max_length": 309.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 297.0, "completions/min_terminated_length": 282.0, "completions/max_terminated_length": 309.0, "rewards/meter/mean": 0.9778913855552673, "rewards/meter/std": 0.025907516479492188, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.03928370773792267, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7741815447807312, "rewards/repeat_penalty/std": 0.09523647278547287, "rewards/total_composite/mean": 0.661956250667572, "rewards/total_composite/std": 0.08313874155282974, "reward": 0.661956250667572, "reward_std": 0.08313874155282974, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027429981157183647, "sampling/sampling_logp_difference/max": 7.558941841125488, "sampling/importance_sampling_ratio/min": 0.0005214267293922603, "sampling/importance_sampling_ratio/mean": 1.0018694400787354, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12555130943655968, "clip_ratio/low_mean": 0.003884511475916952, "clip_ratio/low_min": 0.003884511475916952, "clip_ratio/high_mean": 0.012181702419184148, "clip_ratio/high_max": 0.012181702419184148, "clip_ratio/region_mean": 0.0160662138951011, "reward_total_mean": 0.661956250667572, "reward_meter_mean": 0.9778913855552673, "reward_meter_std": 0.025907516479492188, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.03928370773792267, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7741815447807312, "reward_repeat_penalty_std": 0.09523647278547287, "reward_total_composite_mean": 0.661956250667572, "reward_total_composite_std": 0.08313874155282974} {"timestamp_utc": "2026-04-12T01:06:45Z", "mode": "train", "global_step": 1927, "epoch": 0.07739888339960638, "loss": -0.0006, "grad_norm": 0.516058087348938, "learning_rate": 4.163636363636364e-06, "num_tokens": 4350132.0, "completions/mean_length": 98.875, "completions/min_length": 98.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.875, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.9990072250366211, "rewards/meter/std": 0.00012045173207297921, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990072250366211, "rewards/total_composite/std": 0.00012045173207297921, "reward": 0.9990072250366211, "reward_std": 0.00012044955656165257, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004538480192422867, "sampling/sampling_logp_difference/max": 0.5121603012084961, "sampling/importance_sampling_ratio/min": 0.5991997718811035, "sampling/importance_sampling_ratio/mean": 0.9992603659629822, "sampling/importance_sampling_ratio/max": 1.2373576164245605, "entropy": 0.02114183083176613, "clip_ratio/low_mean": 0.0025252525229007006, "clip_ratio/low_min": 0.0025252525229007006, "clip_ratio/high_mean": 0.0025252525229007006, "clip_ratio/high_max": 0.0025252525229007006, "clip_ratio/region_mean": 0.005050505045801401, "reward_total_mean": 0.9990072250366211, "reward_meter_mean": 0.9990072250366211, "reward_meter_std": 0.00012045173207297921, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990072250366211, "reward_total_composite_std": 0.00012045173207297921} {"timestamp_utc": "2026-04-12T01:06:49Z", "mode": "train", "global_step": 1928, "epoch": 0.07743904888139133, "loss": -0.0147, "grad_norm": 3.781970977783203, "learning_rate": 4.160606060606061e-06, "num_tokens": 4351735.0, "completions/mean_length": 61.375, "completions/min_length": 60.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.375, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9964951276779175, "rewards/meter/std": 0.00043850651127286255, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9964951276779175, "rewards/total_composite/std": 0.00043850651127286255, "reward": 0.9964951276779175, "reward_std": 0.0004385166976135224, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007140871603041887, "sampling/sampling_logp_difference/max": 0.4663825035095215, "sampling/importance_sampling_ratio/min": 0.6272673606872559, "sampling/importance_sampling_ratio/mean": 1.0027474164962769, "sampling/importance_sampling_ratio/max": 1.4405137300491333, "entropy": 0.04211301123723388, "clip_ratio/low_mean": 0.006048386916518211, "clip_ratio/low_min": 0.006048386916518211, "clip_ratio/high_mean": 0.0019841270986944437, "clip_ratio/high_max": 0.0019841270986944437, "clip_ratio/region_mean": 0.008032514015212655, "reward_total_mean": 0.9964951276779175, "reward_meter_mean": 0.9964951276779175, "reward_meter_std": 0.00043850651127286255, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9964951276779175, "reward_total_composite_std": 0.00043850651127286255} {"timestamp_utc": "2026-04-12T01:06:54Z", "mode": "train", "global_step": 1929, "epoch": 0.07747921436317629, "loss": -0.0459, "grad_norm": 3.9287514686584473, "learning_rate": 4.157575757575758e-06, "num_tokens": 4353662.0, "completions/mean_length": 81.875, "completions/min_length": 78.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.875, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9926681518554688, "rewards/meter/std": 0.00178977579344064, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9926681518554688, "rewards/total_composite/std": 0.00178977579344064, "reward": 0.9926681518554688, "reward_std": 0.0017897515790537, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016444914042949677, "sampling/sampling_logp_difference/max": 0.9670681953430176, "sampling/importance_sampling_ratio/min": 0.38019606471061707, "sampling/importance_sampling_ratio/mean": 1.0027321577072144, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07624188321642578, "clip_ratio/low_mean": 0.0126003761542961, "clip_ratio/low_min": 0.0126003761542961, "clip_ratio/high_mean": 0.0013888889225199819, "clip_ratio/high_max": 0.0013888889225199819, "clip_ratio/region_mean": 0.013989265076816082, "reward_total_mean": 0.9926681518554688, "reward_meter_mean": 0.9926681518554688, "reward_meter_std": 0.00178977579344064, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9926681518554688, "reward_total_composite_std": 0.00178977579344064} {"timestamp_utc": "2026-04-12T01:06:59Z", "mode": "train", "global_step": 1930, "epoch": 0.07751937984496124, "loss": 0.0178, "grad_norm": 3.3098363876342773, "learning_rate": 4.154545454545455e-06, "num_tokens": 4355551.0, "completions/mean_length": 67.125, "completions/min_length": 66.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.125, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.84641432762146, "rewards/meter/std": 0.031771186739206314, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.84641432762146, "rewards/total_composite/std": 0.031771186739206314, "reward": 0.84641432762146, "reward_std": 0.03177119046449661, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027442539110779762, "sampling/sampling_logp_difference/max": 1.9472160339355469, "sampling/importance_sampling_ratio/min": 0.14267070591449738, "sampling/importance_sampling_ratio/mean": 1.0048232078552246, "sampling/importance_sampling_ratio/max": 1.4023113250732422, "entropy": 0.19507121294736862, "clip_ratio/low_mean": 0.0035714285913854837, "clip_ratio/low_min": 0.0035714285913854837, "clip_ratio/high_mean": 0.01119648793246597, "clip_ratio/high_max": 0.01119648793246597, "clip_ratio/region_mean": 0.014767916523851454, "reward_total_mean": 0.84641432762146, "reward_meter_mean": 0.84641432762146, "reward_meter_std": 0.031771186739206314, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.84641432762146, "reward_total_composite_std": 0.031771186739206314} {"timestamp_utc": "2026-04-12T01:07:03Z", "mode": "train", "global_step": 1931, "epoch": 0.0775595453267462, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.151515151515152e-06, "num_tokens": 4357311.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.998939037322998, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998939037322998, "rewards/total_composite/std": 0.0, "reward": 0.998939037322998, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0009288773289881647, "sampling/sampling_logp_difference/max": 0.029468350112438202, "sampling/importance_sampling_ratio/min": 0.9709616303443909, "sampling/importance_sampling_ratio/mean": 1.0005451440811157, "sampling/importance_sampling_ratio/max": 1.0244560241699219, "entropy": 0.008458438969682902, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.998939037322998, "reward_meter_mean": 0.998939037322998, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.998939037322998, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:07:08Z", "mode": "train", "global_step": 1932, "epoch": 0.07759971080853115, "loss": 0.0098, "grad_norm": 3.314406156539917, "learning_rate": 4.148484848484849e-06, "num_tokens": 4359128.0, "completions/mean_length": 73.125, "completions/min_length": 70.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.125, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9976757764816284, "rewards/meter/std": 0.0007166875875554979, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9976757764816284, "rewards/total_composite/std": 0.0007166875875554979, "reward": 0.9976757764816284, "reward_std": 0.0007166775758378208, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.029741626232862473, "sampling/sampling_logp_difference/max": 1.3467721939086914, "sampling/importance_sampling_ratio/min": 0.26007840037345886, "sampling/importance_sampling_ratio/mean": 1.001599907875061, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17044041864573956, "clip_ratio/low_mean": 0.0050675676902756095, "clip_ratio/low_min": 0.0050675676902756095, "clip_ratio/high_mean": 0.024061617092229426, "clip_ratio/high_max": 0.024061617092229426, "clip_ratio/region_mean": 0.029129184782505035, "reward_total_mean": 0.9976757764816284, "reward_meter_mean": 0.9976757764816284, "reward_meter_std": 0.0007166875875554979, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9976757764816284, "reward_total_composite_std": 0.0007166875875554979} {"timestamp_utc": "2026-04-12T01:07:13Z", "mode": "train", "global_step": 1933, "epoch": 0.0776398762903161, "loss": 0.0007, "grad_norm": 4.072077751159668, "learning_rate": 4.145454545454546e-06, "num_tokens": 4361033.0, "completions/mean_length": 73.125, "completions/min_length": 71.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.125, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9970039129257202, "rewards/meter/std": 0.0011512924684211612, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9970039129257202, "rewards/total_composite/std": 0.0011512924684211612, "reward": 0.9970039129257202, "reward_std": 0.0011513038771227002, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.029263032600283623, "sampling/sampling_logp_difference/max": 1.3621405363082886, "sampling/importance_sampling_ratio/min": 0.2561119794845581, "sampling/importance_sampling_ratio/mean": 1.0008363723754883, "sampling/importance_sampling_ratio/max": 1.6174455881118774, "entropy": 0.20059540774673223, "clip_ratio/low_mean": 0.008469702675938606, "clip_ratio/low_min": 0.008469702675938606, "clip_ratio/high_mean": 0.0155549660557881, "clip_ratio/high_max": 0.0155549660557881, "clip_ratio/region_mean": 0.024024668731726706, "reward_total_mean": 0.9970039129257202, "reward_meter_mean": 0.9970039129257202, "reward_meter_std": 0.0011512924684211612, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9970039129257202, "reward_total_composite_std": 0.0011512924684211612} {"timestamp_utc": "2026-04-12T01:07:19Z", "mode": "train", "global_step": 1934, "epoch": 0.07768004177210105, "loss": 0.0093, "grad_norm": 3.0962040424346924, "learning_rate": 4.142424242424243e-06, "num_tokens": 4364465.0, "completions/mean_length": 226.0, "completions/min_length": 224.0, "completions/max_length": 230.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 226.0, "completions/min_terminated_length": 224.0, "completions/max_terminated_length": 230.0, "rewards/meter/mean": 0.8571816086769104, "rewards/meter/std": 0.171591117978096, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8020833134651184, "rewards/repeat_penalty/std": 0.0431290864944458, "rewards/total_composite/mean": 0.6912178993225098, "rewards/total_composite/std": 0.158027783036232, "reward": 0.6912178993225098, "reward_std": 0.1580277979373932, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02388615906238556, "sampling/sampling_logp_difference/max": 3.0924410820007324, "sampling/importance_sampling_ratio/min": 0.0453910194337368, "sampling/importance_sampling_ratio/mean": 0.9996902942657471, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10432031005620956, "clip_ratio/low_mean": 0.006601066328585148, "clip_ratio/low_min": 0.006601066328585148, "clip_ratio/high_mean": 0.014982880558818579, "clip_ratio/high_max": 0.014982880558818579, "clip_ratio/region_mean": 0.021583946887403727, "reward_total_mean": 0.6912178993225098, "reward_meter_mean": 0.8571816086769104, "reward_meter_std": 0.171591117978096, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8020833134651184, "reward_repeat_penalty_std": 0.0431290864944458, "reward_total_composite_mean": 0.6912178993225098, "reward_total_composite_std": 0.158027783036232} {"timestamp_utc": "2026-04-12T01:07:24Z", "mode": "train", "global_step": 1935, "epoch": 0.07772020725388601, "loss": -0.0005, "grad_norm": 3.562734365463257, "learning_rate": 4.13939393939394e-06, "num_tokens": 4366373.0, "completions/mean_length": 67.5, "completions/min_length": 66.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.5, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.72654128074646, "rewards/meter/std": 0.2970520853996277, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.72654128074646, "rewards/total_composite/std": 0.2970520853996277, "reward": 0.72654128074646, "reward_std": 0.2970520555973053, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03413822129368782, "sampling/sampling_logp_difference/max": 1.4129490852355957, "sampling/importance_sampling_ratio/min": 0.24342434108257294, "sampling/importance_sampling_ratio/mean": 0.99456787109375, "sampling/importance_sampling_ratio/max": 1.5096417665481567, "entropy": 0.14524980820715427, "clip_ratio/low_mean": 0.007359307492151856, "clip_ratio/low_min": 0.007359307492151856, "clip_ratio/high_mean": 0.025740933255292475, "clip_ratio/high_max": 0.025740933255292475, "clip_ratio/region_mean": 0.03310024074744433, "reward_total_mean": 0.72654128074646, "reward_meter_mean": 0.72654128074646, "reward_meter_std": 0.2970520853996277, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.72654128074646, "reward_total_composite_std": 0.2970520853996277} {"timestamp_utc": "2026-04-12T01:07:29Z", "mode": "train", "global_step": 1936, "epoch": 0.07776037273567096, "loss": -0.0046, "grad_norm": 3.389194965362549, "learning_rate": 4.136363636363637e-06, "num_tokens": 4368431.0, "completions/mean_length": 99.25, "completions/min_length": 98.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.25, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.8136383891105652, "rewards/meter/std": 0.040068093687295914, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.7909245491027832, "rewards/total_composite/std": 0.028373466804623604, "reward": 0.7909245491027832, "reward_std": 0.028373470529913902, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01888117380440235, "sampling/sampling_logp_difference/max": 1.0089292526245117, "sampling/importance_sampling_ratio/min": 0.4062268137931824, "sampling/importance_sampling_ratio/mean": 1.0059889554977417, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15317998873069882, "clip_ratio/low_mean": 0.006351783173158765, "clip_ratio/low_min": 0.006351783173158765, "clip_ratio/high_mean": 0.012465349864214659, "clip_ratio/high_max": 0.012465349864214659, "clip_ratio/region_mean": 0.018817133037373424, "reward_total_mean": 0.7909245491027832, "reward_meter_mean": 0.8136383891105652, "reward_meter_std": 0.040068093687295914, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.7909245491027832, "reward_total_composite_std": 0.028373466804623604} {"timestamp_utc": "2026-04-12T01:07:33Z", "mode": "train", "global_step": 1937, "epoch": 0.07780053821745592, "loss": 0.0008, "grad_norm": 1.4040637016296387, "learning_rate": 4.133333333333333e-06, "num_tokens": 4370119.0, "completions/mean_length": 58.0, "completions/min_length": 58.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.994684100151062, "rewards/meter/std": 5.123758455738425e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.994684100151062, "rewards/total_composite/std": 5.123758455738425e-05, "reward": 0.994684100151062, "reward_std": 5.122838410898112e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004311581142246723, "sampling/sampling_logp_difference/max": 0.4578728675842285, "sampling/importance_sampling_ratio/min": 0.6326279044151306, "sampling/importance_sampling_ratio/mean": 0.9982123374938965, "sampling/importance_sampling_ratio/max": 1.146704077720642, "entropy": 0.018475122982636094, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0021551724057644606, "clip_ratio/high_max": 0.0021551724057644606, "clip_ratio/region_mean": 0.0021551724057644606, "reward_total_mean": 0.994684100151062, "reward_meter_mean": 0.994684100151062, "reward_meter_std": 5.123758455738425e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.994684100151062, "reward_total_composite_std": 5.123758455738425e-05} {"timestamp_utc": "2026-04-12T01:07:38Z", "mode": "train", "global_step": 1938, "epoch": 0.07784070369924087, "loss": -0.0008, "grad_norm": 3.30529522895813, "learning_rate": 4.1303030303030305e-06, "num_tokens": 4372001.0, "completions/mean_length": 67.25, "completions/min_length": 66.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.25, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.8598266839981079, "rewards/meter/std": 0.02622251957654953, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8598266839981079, "rewards/total_composite/std": 0.02622251957654953, "reward": 0.8598266839981079, "reward_std": 0.02622251957654953, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011166814714670181, "sampling/sampling_logp_difference/max": 1.2125158309936523, "sampling/importance_sampling_ratio/min": 0.2974480092525482, "sampling/importance_sampling_ratio/mean": 1.0010411739349365, "sampling/importance_sampling_ratio/max": 1.5219361782073975, "entropy": 0.048515188973397017, "clip_ratio/low_mean": 0.007407813798636198, "clip_ratio/low_min": 0.007407813798636198, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.007407813798636198, "reward_total_mean": 0.8598266839981079, "reward_meter_mean": 0.8598266839981079, "reward_meter_std": 0.02622251957654953, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8598266839981079, "reward_total_composite_std": 0.02622251957654953} {"timestamp_utc": "2026-04-12T01:07:43Z", "mode": "train", "global_step": 1939, "epoch": 0.07788086918102582, "loss": 0.0039, "grad_norm": 3.0253310203552246, "learning_rate": 4.127272727272728e-06, "num_tokens": 4373689.0, "completions/mean_length": 72.0, "completions/min_length": 69.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.0, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.8803695440292358, "rewards/meter/std": 0.31208541989326477, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8803695440292358, "rewards/total_composite/std": 0.31208541989326477, "reward": 0.8803695440292358, "reward_std": 0.3120853900909424, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.035320769995450974, "sampling/sampling_logp_difference/max": 0.9647955894470215, "sampling/importance_sampling_ratio/min": 0.3810611069202423, "sampling/importance_sampling_ratio/mean": 1.0067641735076904, "sampling/importance_sampling_ratio/max": 1.6593668460845947, "entropy": 0.28373553417623043, "clip_ratio/low_mean": 0.0051369862630963326, "clip_ratio/low_min": 0.0051369862630963326, "clip_ratio/high_mean": 0.02452767826616764, "clip_ratio/high_max": 0.02452767826616764, "clip_ratio/region_mean": 0.029664664529263973, "reward_total_mean": 0.8803695440292358, "reward_meter_mean": 0.8803695440292358, "reward_meter_std": 0.31208541989326477, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8803695440292358, "reward_total_composite_std": 0.31208541989326477} {"timestamp_utc": "2026-04-12T01:07:48Z", "mode": "train", "global_step": 1940, "epoch": 0.07792103466281078, "loss": -0.0001, "grad_norm": 0.002891062991693616, "learning_rate": 4.124242424242424e-06, "num_tokens": 4375497.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9989398717880249, "rewards/meter/std": 2.402423206149251e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989398717880249, "rewards/total_composite/std": 2.402423206149251e-06, "reward": 0.9989398717880249, "reward_std": 2.396394847892225e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0017837814521044493, "sampling/sampling_logp_difference/max": 0.4983246326446533, "sampling/importance_sampling_ratio/min": 0.6075477004051208, "sampling/importance_sampling_ratio/mean": 0.9999057054519653, "sampling/importance_sampling_ratio/max": 1.0235236883163452, "entropy": 0.010286982113029808, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0018939394503831863, "clip_ratio/high_max": 0.0018939394503831863, "clip_ratio/region_mean": 0.0018939394503831863, "reward_total_mean": 0.9989398717880249, "reward_meter_mean": 0.9989398717880249, "reward_meter_std": 2.402423206149251e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9989398717880249, "reward_total_composite_std": 2.402423206149251e-06} {"timestamp_utc": "2026-04-12T01:07:58Z", "mode": "train", "global_step": 1941, "epoch": 0.07796120014459573, "loss": -0.3041, "grad_norm": 0.7915775775909424, "learning_rate": 4.1212121212121215e-06, "num_tokens": 4380383.0, "completions/mean_length": 472.75, "completions/min_length": 449.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 467.14288330078125, "completions/min_terminated_length": 449.0, "completions/max_terminated_length": 489.0, "rewards/meter/mean": 0.9804998636245728, "rewards/meter/std": 0.030505068600177765, "rewards/count_adherence/mean": 0.6691176891326904, "rewards/count_adherence/std": 0.062391772866249084, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.9617094993591309, "rewards/repeat_penalty/std": 0.04899247735738754, "rewards/total_composite/mean": 0.5711581707000732, "rewards/total_composite/std": 0.2336382418870926, "reward": 0.5711581707000732, "reward_std": 0.2336382418870926, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.056584302335977554, "sampling/sampling_logp_difference/max": 1.680664300918579, "sampling/importance_sampling_ratio/min": 0.1862502098083496, "sampling/importance_sampling_ratio/mean": 1.0137178897857666, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5811280272901058, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.03567814873531461, "clip_ratio/high_max": 0.03567814873531461, "clip_ratio/region_mean": 0.03567814873531461, "reward_total_mean": 0.5711581707000732, "reward_meter_mean": 0.9804998636245728, "reward_meter_std": 0.030505068600177765, "reward_count_adherence_mean": 0.6691176891326904, "reward_count_adherence_std": 0.062391772866249084, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.9617094993591309, "reward_repeat_penalty_std": 0.04899247735738754, "reward_total_composite_mean": 0.5711581707000732, "reward_total_composite_std": 0.2336382418870926} {"timestamp_utc": "2026-04-12T01:08:02Z", "mode": "train", "global_step": 1942, "epoch": 0.07800136562638069, "loss": -0.008, "grad_norm": 4.029205799102783, "learning_rate": 4.118181818181819e-06, "num_tokens": 4382253.0, "completions/mean_length": 77.75, "completions/min_length": 76.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.75, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9882785081863403, "rewards/meter/std": 0.018005145713686943, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8649736642837524, "rewards/total_composite/std": 0.34996482729911804, "reward": 0.8649736642837524, "reward_std": 0.34996482729911804, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06439422816038132, "sampling/sampling_logp_difference/max": 1.7122077941894531, "sampling/importance_sampling_ratio/min": 0.1804669201374054, "sampling/importance_sampling_ratio/mean": 1.0108014345169067, "sampling/importance_sampling_ratio/max": 1.879008412361145, "entropy": 0.5831749886274338, "clip_ratio/low_mean": 0.004934210330247879, "clip_ratio/low_min": 0.004934210330247879, "clip_ratio/high_mean": 0.03670453419908881, "clip_ratio/high_max": 0.03670453419908881, "clip_ratio/region_mean": 0.04163874452933669, "reward_total_mean": 0.8649736642837524, "reward_meter_mean": 0.9882785081863403, "reward_meter_std": 0.018005145713686943, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8649736642837524, "reward_total_composite_std": 0.34996482729911804} {"timestamp_utc": "2026-04-12T01:08:07Z", "mode": "train", "global_step": 1943, "epoch": 0.07804153110816564, "loss": 0.0281, "grad_norm": 2.609006643295288, "learning_rate": 4.115151515151515e-06, "num_tokens": 4384317.0, "completions/mean_length": 96.0, "completions/min_length": 95.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.0, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.9970648288726807, "rewards/meter/std": 0.0006929152878001332, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9721810817718506, "rewards/total_composite/std": 0.07107478380203247, "reward": 0.9721810817718506, "reward_std": 0.07107476890087128, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00461819302290678, "sampling/sampling_logp_difference/max": 0.3813199996948242, "sampling/importance_sampling_ratio/min": 0.6829593181610107, "sampling/importance_sampling_ratio/mean": 1.0022590160369873, "sampling/importance_sampling_ratio/max": 1.3738577365875244, "entropy": 0.030685836798511446, "clip_ratio/low_mean": 0.0012135922443121672, "clip_ratio/low_min": 0.0012135922443121672, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0012135922443121672, "reward_total_mean": 0.9721810817718506, "reward_meter_mean": 0.9970648288726807, "reward_meter_std": 0.0006929152878001332, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9721810817718506, "reward_total_composite_std": 0.07107478380203247} {"timestamp_utc": "2026-04-12T01:08:12Z", "mode": "train", "global_step": 1944, "epoch": 0.0780816965899506, "loss": -0.0001, "grad_norm": 0.053728338330984116, "learning_rate": 4.112121212121212e-06, "num_tokens": 4385917.0, "completions/mean_length": 63.0, "completions/min_length": 63.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.0, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9970229864120483, "rewards/meter/std": 8.450447239738423e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9970229864120483, "rewards/total_composite/std": 8.450447239738423e-06, "reward": 0.9970229864120483, "reward_std": 8.441417776339222e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0021621037740260363, "sampling/sampling_logp_difference/max": 0.1747570037841797, "sampling/importance_sampling_ratio/min": 0.8396610021591187, "sampling/importance_sampling_ratio/mean": 1.0014358758926392, "sampling/importance_sampling_ratio/max": 1.081364393234253, "entropy": 0.015167628531344235, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0019841270986944437, "clip_ratio/high_max": 0.0019841270986944437, "clip_ratio/region_mean": 0.0019841270986944437, "reward_total_mean": 0.9970229864120483, "reward_meter_mean": 0.9970229864120483, "reward_meter_std": 8.450447239738423e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9970229864120483, "reward_total_composite_std": 8.450447239738423e-06} {"timestamp_utc": "2026-04-12T01:08:16Z", "mode": "train", "global_step": 1945, "epoch": 0.07812186207173555, "loss": -0.0001, "grad_norm": 0.013028639368712902, "learning_rate": 4.10909090909091e-06, "num_tokens": 4387733.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9989398717880249, "rewards/meter/std": 2.402423206149251e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989398717880249, "rewards/total_composite/std": 2.402423206149251e-06, "reward": 0.9989398717880249, "reward_std": 2.396394847892225e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.002988132182508707, "sampling/sampling_logp_difference/max": 0.1946016550064087, "sampling/importance_sampling_ratio/min": 0.8231624960899353, "sampling/importance_sampling_ratio/mean": 1.0004053115844727, "sampling/importance_sampling_ratio/max": 1.1326637268066406, "entropy": 0.019811521749943495, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0018939394503831863, "clip_ratio/high_max": 0.0018939394503831863, "clip_ratio/region_mean": 0.0018939394503831863, "reward_total_mean": 0.9989398717880249, "reward_meter_mean": 0.9989398717880249, "reward_meter_std": 2.402423206149251e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9989398717880249, "reward_total_composite_std": 2.402423206149251e-06} {"timestamp_utc": "2026-04-12T01:08:21Z", "mode": "train", "global_step": 1946, "epoch": 0.0781620275535205, "loss": 0.0117, "grad_norm": 2.780980110168457, "learning_rate": 4.106060606060606e-06, "num_tokens": 4389585.0, "completions/mean_length": 67.5, "completions/min_length": 66.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.5, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.8567509055137634, "rewards/meter/std": 0.07118333131074905, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8567509055137634, "rewards/total_composite/std": 0.07118333131074905, "reward": 0.8567509055137634, "reward_std": 0.07118332386016846, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019841060042381287, "sampling/sampling_logp_difference/max": 1.137892723083496, "sampling/importance_sampling_ratio/min": 0.3204936981201172, "sampling/importance_sampling_ratio/mean": 1.0084105730056763, "sampling/importance_sampling_ratio/max": 1.5270895957946777, "entropy": 0.18513774313032627, "clip_ratio/low_mean": 0.003623949596658349, "clip_ratio/low_min": 0.003623949596658349, "clip_ratio/high_mean": 0.011114974273368716, "clip_ratio/high_max": 0.011114974273368716, "clip_ratio/region_mean": 0.014738923870027065, "reward_total_mean": 0.8567509055137634, "reward_meter_mean": 0.8567509055137634, "reward_meter_std": 0.07118333131074905, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8567509055137634, "reward_total_composite_std": 0.07118333131074905} {"timestamp_utc": "2026-04-12T01:08:26Z", "mode": "train", "global_step": 1947, "epoch": 0.07820219303530546, "loss": 0.003, "grad_norm": 2.5665481090545654, "learning_rate": 4.103030303030303e-06, "num_tokens": 4391466.0, "completions/mean_length": 73.125, "completions/min_length": 73.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.125, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9979450702667236, "rewards/meter/std": 0.00038199007394723594, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979450702667236, "rewards/total_composite/std": 0.00038199007394723594, "reward": 0.9979450702667236, "reward_std": 0.00038197485264390707, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02320614643394947, "sampling/sampling_logp_difference/max": 1.1087846755981445, "sampling/importance_sampling_ratio/min": 0.3299597203731537, "sampling/importance_sampling_ratio/mean": 0.9995145201683044, "sampling/importance_sampling_ratio/max": 1.5535597801208496, "entropy": 0.1381580764427781, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.011963161756284535, "clip_ratio/high_max": 0.011963161756284535, "clip_ratio/region_mean": 0.011963161756284535, "reward_total_mean": 0.9979450702667236, "reward_meter_mean": 0.9979450702667236, "reward_meter_std": 0.00038199007394723594, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9979450702667236, "reward_total_composite_std": 0.00038199007394723594} {"timestamp_utc": "2026-04-12T01:08:31Z", "mode": "train", "global_step": 1948, "epoch": 0.07824235851709041, "loss": 0.0029, "grad_norm": 2.1724514961242676, "learning_rate": 4.1e-06, "num_tokens": 4393381.0, "completions/mean_length": 73.375, "completions/min_length": 72.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.375, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9980074167251587, "rewards/meter/std": 0.0006349986069835722, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980074167251587, "rewards/total_composite/std": 0.0006349986069835722, "reward": 0.9980074167251587, "reward_std": 0.0006350158364512026, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025800568982958794, "sampling/sampling_logp_difference/max": 0.789684534072876, "sampling/importance_sampling_ratio/min": 0.4539879858493805, "sampling/importance_sampling_ratio/mean": 1.0009372234344482, "sampling/importance_sampling_ratio/max": 1.8275814056396484, "entropy": 0.16229457408189774, "clip_ratio/low_mean": 0.00856354646384716, "clip_ratio/low_min": 0.00856354646384716, "clip_ratio/high_mean": 0.012034508981741965, "clip_ratio/high_max": 0.012034508981741965, "clip_ratio/region_mean": 0.020598055445589125, "reward_total_mean": 0.9980074167251587, "reward_meter_mean": 0.9980074167251587, "reward_meter_std": 0.0006349986069835722, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9980074167251587, "reward_total_composite_std": 0.0006349986069835722} {"timestamp_utc": "2026-04-12T01:08:36Z", "mode": "train", "global_step": 1949, "epoch": 0.07828252399887536, "loss": -0.0289, "grad_norm": 5.015791893005371, "learning_rate": 4.096969696969697e-06, "num_tokens": 4395881.0, "completions/mean_length": 122.5, "completions/min_length": 110.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.5, "completions/min_terminated_length": 110.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.10345004498958588, "rewards/meter/std": 0.17283838987350464, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.851190447807312, "rewards/repeat_penalty/std": 0.08266931027173996, "rewards/total_composite/mean": 0.0730920359492302, "rewards/total_composite/std": 0.11985667049884796, "reward": 0.0730920359492302, "reward_std": 0.11985667049884796, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03781670331954956, "sampling/sampling_logp_difference/max": 1.3046746253967285, "sampling/importance_sampling_ratio/min": 0.2712607979774475, "sampling/importance_sampling_ratio/mean": 0.9997954368591309, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1550168227404356, "clip_ratio/low_mean": 0.023918771417811513, "clip_ratio/low_min": 0.023918771417811513, "clip_ratio/high_mean": 0.012344544753432274, "clip_ratio/high_max": 0.012344544753432274, "clip_ratio/region_mean": 0.03626331617124379, "reward_total_mean": 0.0730920359492302, "reward_meter_mean": 0.10345004498958588, "reward_meter_std": 0.17283838987350464, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.851190447807312, "reward_repeat_penalty_std": 0.08266931027173996, "reward_total_composite_mean": 0.0730920359492302, "reward_total_composite_std": 0.11985667049884796} {"timestamp_utc": "2026-04-12T01:08:42Z", "mode": "train", "global_step": 1950, "epoch": 0.07832268948066032, "loss": 0.3165, "grad_norm": 6.135134220123291, "learning_rate": 4.093939393939394e-06, "num_tokens": 4397640.0, "completions/mean_length": 54.875, "completions/min_length": 37.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.875, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9981762170791626, "rewards/meter/std": 0.0004922283114865422, "rewards/count_adherence/mean": 0.5, "rewards/count_adherence/std": 0.5345224738121033, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.49901658296585083, "rewards/total_composite/std": 0.5334713459014893, "reward": 0.49901658296585083, "reward_std": 0.5334712862968445, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02422521449625492, "sampling/sampling_logp_difference/max": 1.499140977859497, "sampling/importance_sampling_ratio/min": 0.22332191467285156, "sampling/importance_sampling_ratio/mean": 1.0079329013824463, "sampling/importance_sampling_ratio/max": 1.6159636974334717, "entropy": 0.1553973164409399, "clip_ratio/low_mean": 0.00868320802692324, "clip_ratio/low_min": 0.00868320802692324, "clip_ratio/high_mean": 0.0033783784601837397, "clip_ratio/high_max": 0.0033783784601837397, "clip_ratio/region_mean": 0.012061586487106979, "reward_total_mean": 0.49901658296585083, "reward_meter_mean": 0.9981762170791626, "reward_meter_std": 0.0004922283114865422, "reward_count_adherence_mean": 0.5, "reward_count_adherence_std": 0.5345224738121033, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.49901658296585083, "reward_total_composite_std": 0.5334713459014893} {"timestamp_utc": "2026-04-12T01:09:56Z", "mode": "eval", "global_step": 1950, "epoch": 0.07832268948066032, "eval_loss": NaN, "eval_runtime": 74.1396, "eval_samples_per_second": 1.403, "eval_steps_per_second": 0.175, "eval_num_tokens": 4397640.0, "eval_completions/mean_length": 219.20192307692307, "eval_completions/min_length": 67.84615384615384, "eval_completions/max_length": 383.15384615384613, "eval_completions/clipped_ratio": 0.009615384615384616, "eval_completions/mean_terminated_length": 216.13324209359976, "eval_completions/min_terminated_length": 67.84615384615384, "eval_completions/max_terminated_length": 372.84615384615387, "eval_rewards/meter/mean": 0.7193602369381831, "eval_rewards/meter/std": 0.41060519218444824, "eval_rewards/count_adherence/mean": 0.8801831007003784, "eval_rewards/count_adherence/std": 0.15413257680260217, "eval_rewards/arabic_clean/mean": 0.9807692307692307, "eval_rewards/arabic_clean/std": 0.05439282839114849, "eval_rewards/repeat_penalty/mean": 0.8391693326143118, "eval_rewards/repeat_penalty/std": 0.14583237583820635, "eval_rewards/total_composite/mean": 0.5268221841408656, "eval_rewards/total_composite/std": 0.36114954260679394, "eval_reward": 0.5268221841408656, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.02039779959103236, "eval_sampling/sampling_logp_difference/max": 1.0237883787888746, "eval_sampling/importance_sampling_ratio/min": 0.3655441105365753, "eval_sampling/importance_sampling_ratio/mean": 1.0057256405170147, "eval_sampling/importance_sampling_ratio/max": 1.5754675223277166, "eval_entropy": 0.2521360000738731, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5268221841408656, "eval_reward_meter_mean": 0.7193602369381831, "eval_reward_meter_std": 0.41060519218444824, "eval_reward_count_adherence_mean": 0.8801831007003784, "eval_reward_count_adherence_std": 0.15413257680260217, "eval_reward_arabic_clean_mean": 0.9807692307692307, "eval_reward_arabic_clean_std": 0.05439282839114849, "eval_reward_repeat_penalty_mean": 0.8391693326143118, "eval_reward_repeat_penalty_std": 0.14583237583820635, "eval_reward_total_composite_mean": 0.5268221841408656, "eval_reward_total_composite_std": 0.36114954260679394} {"timestamp_utc": "2026-04-12T01:10:04Z", "mode": "train", "global_step": 1951, "epoch": 0.07836285496244527, "loss": 0.0003, "grad_norm": 0.36703649163246155, "learning_rate": 4.0909090909090915e-06, "num_tokens": 4399744.0, "completions/mean_length": 90.0, "completions/min_length": 90.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.0, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9955518841743469, "rewards/meter/std": 1.7999940610025078e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9955518841743469, "rewards/total_composite/std": 1.7999940610025078e-05, "reward": 0.9955518841743469, "reward_std": 1.8004628145718016e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004722072742879391, "sampling/sampling_logp_difference/max": 0.8973627090454102, "sampling/importance_sampling_ratio/min": 0.40764331817626953, "sampling/importance_sampling_ratio/mean": 0.9992460012435913, "sampling/importance_sampling_ratio/max": 1.191457748413086, "entropy": 0.015661518438719213, "clip_ratio/low_mean": 0.0027777778450399637, "clip_ratio/low_min": 0.0027777778450399637, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0027777778450399637, "reward_total_mean": 0.9955518841743469, "reward_meter_mean": 0.9955518841743469, "reward_meter_std": 1.7999940610025078e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9955518841743469, "reward_total_composite_std": 1.7999940610025078e-05} {"timestamp_utc": "2026-04-12T01:10:10Z", "mode": "train", "global_step": 1952, "epoch": 0.07840302044423023, "loss": -0.0014, "grad_norm": 4.403555870056152, "learning_rate": 4.087878787878789e-06, "num_tokens": 4401615.0, "completions/mean_length": 71.875, "completions/min_length": 70.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.875, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.987866997718811, "rewards/meter/std": 0.015270087867975235, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.987866997718811, "rewards/total_composite/std": 0.015270087867975235, "reward": 0.987866997718811, "reward_std": 0.015270096249878407, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028592025861144066, "sampling/sampling_logp_difference/max": 0.8273954391479492, "sampling/importance_sampling_ratio/min": 0.43718650937080383, "sampling/importance_sampling_ratio/mean": 1.002774715423584, "sampling/importance_sampling_ratio/max": 1.4481229782104492, "entropy": 0.23499170877039433, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.031067086150869727, "clip_ratio/high_max": 0.031067086150869727, "clip_ratio/region_mean": 0.031067086150869727, "reward_total_mean": 0.987866997718811, "reward_meter_mean": 0.987866997718811, "reward_meter_std": 0.015270087867975235, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.987866997718811, "reward_total_composite_std": 0.015270087867975235} {"timestamp_utc": "2026-04-12T01:10:15Z", "mode": "train", "global_step": 1953, "epoch": 0.07844318592601518, "loss": 0.0586, "grad_norm": 6.583865165710449, "learning_rate": 4.084848484848485e-06, "num_tokens": 4403483.0, "completions/mean_length": 74.5, "completions/min_length": 59.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.5, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.43240049481391907, "rewards/meter/std": 0.30809712409973145, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.41915929317474365, "rewards/total_composite/std": 0.30558672547340393, "reward": 0.41915929317474365, "reward_std": 0.30558669567108154, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.043784063309431076, "sampling/sampling_logp_difference/max": 1.1320176124572754, "sampling/importance_sampling_ratio/min": 0.3223821520805359, "sampling/importance_sampling_ratio/mean": 1.0041871070861816, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2592748161405325, "clip_ratio/low_mean": 0.012907008291222155, "clip_ratio/low_min": 0.012907008291222155, "clip_ratio/high_mean": 0.029118790524080396, "clip_ratio/high_max": 0.029118790524080396, "clip_ratio/region_mean": 0.04202579881530255, "reward_total_mean": 0.41915929317474365, "reward_meter_mean": 0.43240049481391907, "reward_meter_std": 0.30809712409973145, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.41915929317474365, "reward_total_composite_std": 0.30558672547340393} {"timestamp_utc": "2026-04-12T01:10:20Z", "mode": "train", "global_step": 1954, "epoch": 0.07848335140780013, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.081818181818182e-06, "num_tokens": 4405203.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9989815950393677, "rewards/meter/std": 5.247284934739582e-05, "rewards/count_adherence/mean": 0.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.0, "rewards/total_composite/std": 0.0, "reward": 0.0, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.007122131530195475, "sampling/sampling_logp_difference/max": 0.5082035064697266, "sampling/importance_sampling_ratio/min": 0.6015753746032715, "sampling/importance_sampling_ratio/mean": 1.001094937324524, "sampling/importance_sampling_ratio/max": 1.501068353652954, "entropy": 0.02731157885864377, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.0, "reward_meter_mean": 0.9989815950393677, "reward_meter_std": 5.247284934739582e-05, "reward_count_adherence_mean": 0.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.0, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:10:28Z", "mode": "train", "global_step": 1955, "epoch": 0.07852351688958509, "loss": -0.0135, "grad_norm": 1.9077023267745972, "learning_rate": 4.07878787878788e-06, "num_tokens": 4409222.0, "completions/mean_length": 300.375, "completions/min_length": 293.0, "completions/max_length": 315.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 300.375, "completions/min_terminated_length": 293.0, "completions/max_terminated_length": 315.0, "rewards/meter/mean": 0.9946515560150146, "rewards/meter/std": 0.002225227653980255, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7741013169288635, "rewards/repeat_penalty/std": 0.08203362673521042, "rewards/total_composite/mean": 0.6738342642784119, "rewards/total_composite/std": 0.07273107022047043, "reward": 0.6738342642784119, "reward_std": 0.07273106276988983, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018673671409487724, "sampling/sampling_logp_difference/max": 2.512004852294922, "sampling/importance_sampling_ratio/min": 0.08110546320676804, "sampling/importance_sampling_ratio/mean": 1.0040055513381958, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12408511992543936, "clip_ratio/low_mean": 0.007172186364186928, "clip_ratio/low_min": 0.007172186364186928, "clip_ratio/high_mean": 0.005274800234474242, "clip_ratio/high_max": 0.005274800234474242, "clip_ratio/region_mean": 0.01244698659866117, "reward_total_mean": 0.6738342642784119, "reward_meter_mean": 0.9946515560150146, "reward_meter_std": 0.002225227653980255, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7741013169288635, "reward_repeat_penalty_std": 0.08203362673521042, "reward_total_composite_mean": 0.6738342642784119, "reward_total_composite_std": 0.07273107022047043} {"timestamp_utc": "2026-04-12T01:10:32Z", "mode": "train", "global_step": 1956, "epoch": 0.07856368237137004, "loss": 0.0496, "grad_norm": 8.814888000488281, "learning_rate": 4.075757575757576e-06, "num_tokens": 4410857.0, "completions/mean_length": 47.375, "completions/min_length": 43.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.375, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.6137043237686157, "rewards/meter/std": 0.33018791675567627, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6137043237686157, "rewards/total_composite/std": 0.33018791675567627, "reward": 0.6137043237686157, "reward_std": 0.33018791675567627, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06914983689785004, "sampling/sampling_logp_difference/max": 1.6852831840515137, "sampling/importance_sampling_ratio/min": 0.18539191782474518, "sampling/importance_sampling_ratio/mean": 0.9924876689910889, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3848374430090189, "clip_ratio/low_mean": 0.028407905949279666, "clip_ratio/low_min": 0.028407905949279666, "clip_ratio/high_mean": 0.041018035262823105, "clip_ratio/high_max": 0.041018035262823105, "clip_ratio/region_mean": 0.06942594121210277, "reward_total_mean": 0.6137043237686157, "reward_meter_mean": 0.6137043237686157, "reward_meter_std": 0.33018791675567627, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6137043237686157, "reward_total_composite_std": 0.33018791675567627} {"timestamp_utc": "2026-04-12T01:10:38Z", "mode": "train", "global_step": 1957, "epoch": 0.078603847853155, "loss": 0.0042, "grad_norm": 2.345047950744629, "learning_rate": 4.072727272727273e-06, "num_tokens": 4413295.0, "completions/mean_length": 136.75, "completions/min_length": 136.0, "completions/max_length": 139.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 136.75, "completions/min_terminated_length": 136.0, "completions/max_terminated_length": 139.0, "rewards/meter/mean": 0.9962515830993652, "rewards/meter/std": 0.0024396260268986225, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.8717420101165771, "rewards/total_composite/std": 0.05084509775042534, "reward": 0.8717420101165771, "reward_std": 0.05084508657455444, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01802092418074608, "sampling/sampling_logp_difference/max": 3.096953868865967, "sampling/importance_sampling_ratio/min": 0.045186638832092285, "sampling/importance_sampling_ratio/mean": 1.0027059316635132, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0967075265944004, "clip_ratio/low_mean": 0.009184467780869454, "clip_ratio/low_min": 0.009184467780869454, "clip_ratio/high_mean": 0.0018382353009656072, "clip_ratio/high_max": 0.0018382353009656072, "clip_ratio/region_mean": 0.011022703081835061, "reward_total_mean": 0.8717420101165771, "reward_meter_mean": 0.9962515830993652, "reward_meter_std": 0.0024396260268986225, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.8717420101165771, "reward_total_composite_std": 0.05084509775042534} {"timestamp_utc": "2026-04-12T01:10:42Z", "mode": "train", "global_step": 1958, "epoch": 0.07864401333493995, "loss": 0.0012, "grad_norm": 10.1416654586792, "learning_rate": 4.0696969696969706e-06, "num_tokens": 4414671.0, "completions/mean_length": 33.0, "completions/min_length": 33.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9584056735038757, "rewards/meter/std": 0.020386451855301857, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9584056735038757, "rewards/total_composite/std": 0.020386451855301857, "reward": 0.9584056735038757, "reward_std": 0.020386451855301857, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025948278605937958, "sampling/sampling_logp_difference/max": 0.9829981327056885, "sampling/importance_sampling_ratio/min": 0.37418755888938904, "sampling/importance_sampling_ratio/mean": 1.0064359903335571, "sampling/importance_sampling_ratio/max": 1.6090956926345825, "entropy": 0.20033110305666924, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.007575757801532745, "clip_ratio/high_max": 0.007575757801532745, "clip_ratio/region_mean": 0.007575757801532745, "reward_total_mean": 0.9584056735038757, "reward_meter_mean": 0.9584056735038757, "reward_meter_std": 0.020386451855301857, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9584056735038757, "reward_total_composite_std": 0.020386451855301857} {"timestamp_utc": "2026-04-12T01:10:47Z", "mode": "train", "global_step": 1959, "epoch": 0.0786841788167249, "loss": -0.0107, "grad_norm": 8.695691108703613, "learning_rate": 4.066666666666667e-06, "num_tokens": 4416753.0, "completions/mean_length": 76.25, "completions/min_length": 70.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.25, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9713230133056641, "rewards/meter/std": 0.06465231627225876, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9713230133056641, "rewards/total_composite/std": 0.06465231627225876, "reward": 0.9713230133056641, "reward_std": 0.06465231627225876, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04479827731847763, "sampling/sampling_logp_difference/max": 1.1926956176757812, "sampling/importance_sampling_ratio/min": 0.303402304649353, "sampling/importance_sampling_ratio/mean": 1.0147606134414673, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.44979994744062424, "clip_ratio/low_mean": 0.0035714285913854837, "clip_ratio/low_min": 0.0035714285913854837, "clip_ratio/high_mean": 0.028867413057014346, "clip_ratio/high_max": 0.028867413057014346, "clip_ratio/region_mean": 0.03243884164839983, "reward_total_mean": 0.9713230133056641, "reward_meter_mean": 0.9713230133056641, "reward_meter_std": 0.06465231627225876, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9713230133056641, "reward_total_composite_std": 0.06465231627225876} {"timestamp_utc": "2026-04-12T01:10:51Z", "mode": "train", "global_step": 1960, "epoch": 0.07872434429850986, "loss": -0.0106, "grad_norm": 5.968605041503906, "learning_rate": 4.063636363636364e-06, "num_tokens": 4418613.0, "completions/mean_length": 66.5, "completions/min_length": 64.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.5, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.7400355339050293, "rewards/meter/std": 0.26013627648353577, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7400355339050293, "rewards/total_composite/std": 0.26013627648353577, "reward": 0.7400355339050293, "reward_std": 0.26013627648353577, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03610033914446831, "sampling/sampling_logp_difference/max": 1.647068977355957, "sampling/importance_sampling_ratio/min": 0.1926136314868927, "sampling/importance_sampling_ratio/mean": 1.0059431791305542, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.253327926620841, "clip_ratio/low_mean": 0.007722355891019106, "clip_ratio/low_min": 0.007722355891019106, "clip_ratio/high_mean": 0.0037313431967049837, "clip_ratio/high_max": 0.0037313431967049837, "clip_ratio/region_mean": 0.01145369908772409, "reward_total_mean": 0.7400355339050293, "reward_meter_mean": 0.7400355339050293, "reward_meter_std": 0.26013627648353577, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7400355339050293, "reward_total_composite_std": 0.26013627648353577} {"timestamp_utc": "2026-04-12T01:10:56Z", "mode": "train", "global_step": 1961, "epoch": 0.07876450978029481, "loss": -0.0019, "grad_norm": 0.6822566390037537, "learning_rate": 4.060606060606061e-06, "num_tokens": 4420309.0, "completions/mean_length": 63.0, "completions/min_length": 63.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.0, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.996955156326294, "rewards/meter/std": 0.0001835073926486075, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.996955156326294, "rewards/total_composite/std": 0.0001835073926486075, "reward": 0.996955156326294, "reward_std": 0.00018350737809669226, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003504997817799449, "sampling/sampling_logp_difference/max": 0.9989252090454102, "sampling/importance_sampling_ratio/min": 0.36827507615089417, "sampling/importance_sampling_ratio/mean": 1.0001448392868042, "sampling/importance_sampling_ratio/max": 1.0358319282531738, "entropy": 0.014339913497678936, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.996955156326294, "reward_meter_mean": 0.996955156326294, "reward_meter_std": 0.0001835073926486075, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.996955156326294, "reward_total_composite_std": 0.0001835073926486075} {"timestamp_utc": "2026-04-12T01:11:01Z", "mode": "train", "global_step": 1962, "epoch": 0.07880467526207977, "loss": 0.0256, "grad_norm": 3.0988998413085938, "learning_rate": 4.057575757575758e-06, "num_tokens": 4422201.0, "completions/mean_length": 77.5, "completions/min_length": 73.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.5, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9931849837303162, "rewards/meter/std": 0.005035399924963713, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9931849837303162, "rewards/total_composite/std": 0.005035399924963713, "reward": 0.9931849837303162, "reward_std": 0.00503541948273778, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04507505148649216, "sampling/sampling_logp_difference/max": 1.2003469467163086, "sampling/importance_sampling_ratio/min": 0.30108973383903503, "sampling/importance_sampling_ratio/mean": 1.0084387063980103, "sampling/importance_sampling_ratio/max": 1.963693380355835, "entropy": 0.5117241851985455, "clip_ratio/low_mean": 0.006172839552164078, "clip_ratio/low_min": 0.006172839552164078, "clip_ratio/high_mean": 0.01814535935409367, "clip_ratio/high_max": 0.01814535935409367, "clip_ratio/region_mean": 0.02431819890625775, "reward_total_mean": 0.9931849837303162, "reward_meter_mean": 0.9931849837303162, "reward_meter_std": 0.005035399924963713, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9931849837303162, "reward_total_composite_std": 0.005035399924963713} {"timestamp_utc": "2026-04-12T01:11:06Z", "mode": "train", "global_step": 1963, "epoch": 0.07884484074386472, "loss": -0.01, "grad_norm": 5.142487049102783, "learning_rate": 4.054545454545455e-06, "num_tokens": 4423998.0, "completions/mean_length": 72.625, "completions/min_length": 71.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.625, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.997757077217102, "rewards/meter/std": 0.0013475378509610891, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.997757077217102, "rewards/total_composite/std": 0.0013475378509610891, "reward": 0.997757077217102, "reward_std": 0.0013475335435941815, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0390477180480957, "sampling/sampling_logp_difference/max": 1.0736570358276367, "sampling/importance_sampling_ratio/min": 0.3417564034461975, "sampling/importance_sampling_ratio/mean": 0.9945013523101807, "sampling/importance_sampling_ratio/max": 1.6228265762329102, "entropy": 0.245115103200078, "clip_ratio/low_mean": 0.0034722222480922937, "clip_ratio/low_min": 0.0034722222480922937, "clip_ratio/high_mean": 0.018698629923164845, "clip_ratio/high_max": 0.018698629923164845, "clip_ratio/region_mean": 0.022170852171257138, "reward_total_mean": 0.997757077217102, "reward_meter_mean": 0.997757077217102, "reward_meter_std": 0.0013475378509610891, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.997757077217102, "reward_total_composite_std": 0.0013475378509610891} {"timestamp_utc": "2026-04-12T01:11:12Z", "mode": "train", "global_step": 1964, "epoch": 0.07888500622564967, "loss": 0.0047, "grad_norm": 2.332364320755005, "learning_rate": 4.0515151515151516e-06, "num_tokens": 4426910.0, "completions/mean_length": 181.0, "completions/min_length": 180.0, "completions/max_length": 183.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 181.0, "completions/min_terminated_length": 180.0, "completions/max_terminated_length": 183.0, "rewards/meter/mean": 0.9980403780937195, "rewards/meter/std": 0.00041446235263720155, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8472222685813904, "rewards/repeat_penalty/std": 0.08266931027173996, "rewards/total_composite/mean": 0.8455538749694824, "rewards/total_composite/std": 0.08242159336805344, "reward": 0.8455538749694824, "reward_std": 0.08242160081863403, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021872445940971375, "sampling/sampling_logp_difference/max": 1.738257884979248, "sampling/importance_sampling_ratio/min": 0.17582646012306213, "sampling/importance_sampling_ratio/mean": 1.0034029483795166, "sampling/importance_sampling_ratio/max": 1.5945336818695068, "entropy": 0.1502722892910242, "clip_ratio/low_mean": 0.001377504551783204, "clip_ratio/low_min": 0.001377504551783204, "clip_ratio/high_mean": 0.009661172516644001, "clip_ratio/high_max": 0.009661172516644001, "clip_ratio/region_mean": 0.011038677068427205, "reward_total_mean": 0.8455538749694824, "reward_meter_mean": 0.9980403780937195, "reward_meter_std": 0.00041446235263720155, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8472222685813904, "reward_repeat_penalty_std": 0.08266931027173996, "reward_total_composite_mean": 0.8455538749694824, "reward_total_composite_std": 0.08242159336805344} {"timestamp_utc": "2026-04-12T01:11:17Z", "mode": "train", "global_step": 1965, "epoch": 0.07892517170743463, "loss": -0.0106, "grad_norm": 7.793933391571045, "learning_rate": 4.048484848484849e-06, "num_tokens": 4428746.0, "completions/mean_length": 74.5, "completions/min_length": 71.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.5, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9973430037498474, "rewards/meter/std": 0.001286112586967647, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973430037498474, "rewards/total_composite/std": 0.001286112586967647, "reward": 0.9973430037498474, "reward_std": 0.0012860975693911314, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05123698338866234, "sampling/sampling_logp_difference/max": 2.1309890747070312, "sampling/importance_sampling_ratio/min": 0.11871980875730515, "sampling/importance_sampling_ratio/mean": 1.0060445070266724, "sampling/importance_sampling_ratio/max": 1.865578293800354, "entropy": 0.43255916610360146, "clip_ratio/low_mean": 0.007042253389954567, "clip_ratio/low_min": 0.007042253389954567, "clip_ratio/high_mean": 0.024971136124804616, "clip_ratio/high_max": 0.024971136124804616, "clip_ratio/region_mean": 0.03201338951475918, "reward_total_mean": 0.9973430037498474, "reward_meter_mean": 0.9973430037498474, "reward_meter_std": 0.001286112586967647, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9973430037498474, "reward_total_composite_std": 0.001286112586967647} {"timestamp_utc": "2026-04-12T01:11:22Z", "mode": "train", "global_step": 1966, "epoch": 0.07896533718921958, "loss": 0.0066, "grad_norm": 2.1039681434631348, "learning_rate": 4.045454545454546e-06, "num_tokens": 4430956.0, "completions/mean_length": 109.25, "completions/min_length": 103.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 109.25, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.993117094039917, "rewards/meter/std": 0.010319688357412815, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9682018160820007, "rewards/total_composite/std": 0.06981157511472702, "reward": 0.9682018160820007, "reward_std": 0.06981158256530762, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04413502663373947, "sampling/sampling_logp_difference/max": 1.36668062210083, "sampling/importance_sampling_ratio/min": 0.2549518346786499, "sampling/importance_sampling_ratio/mean": 1.0064747333526611, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.38045662455260754, "clip_ratio/low_mean": 0.00909090880304575, "clip_ratio/low_min": 0.00909090880304575, "clip_ratio/high_mean": 0.027550982777029276, "clip_ratio/high_max": 0.027550982777029276, "clip_ratio/region_mean": 0.036641891580075026, "reward_total_mean": 0.9682018160820007, "reward_meter_mean": 0.993117094039917, "reward_meter_std": 0.010319688357412815, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9682018160820007, "reward_total_composite_std": 0.06981157511472702} {"timestamp_utc": "2026-04-12T01:11:27Z", "mode": "train", "global_step": 1967, "epoch": 0.07900550267100453, "loss": 0.0108, "grad_norm": 2.3863680362701416, "learning_rate": 4.0424242424242425e-06, "num_tokens": 4432773.0, "completions/mean_length": 71.125, "completions/min_length": 68.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.125, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9883999824523926, "rewards/meter/std": 0.005968651734292507, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9883999824523926, "rewards/total_composite/std": 0.005968651734292507, "reward": 0.9883999824523926, "reward_std": 0.005968641955405474, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04225626587867737, "sampling/sampling_logp_difference/max": 1.4220304489135742, "sampling/importance_sampling_ratio/min": 0.24122373759746552, "sampling/importance_sampling_ratio/mean": 1.0037516355514526, "sampling/importance_sampling_ratio/max": 1.7200582027435303, "entropy": 0.2595768254250288, "clip_ratio/low_mean": 0.022320898715406656, "clip_ratio/low_min": 0.022320898715406656, "clip_ratio/high_mean": 0.019368293695151806, "clip_ratio/high_max": 0.019368293695151806, "clip_ratio/region_mean": 0.04168919241055846, "reward_total_mean": 0.9883999824523926, "reward_meter_mean": 0.9883999824523926, "reward_meter_std": 0.005968651734292507, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9883999824523926, "reward_total_composite_std": 0.005968651734292507} {"timestamp_utc": "2026-04-12T01:11:31Z", "mode": "train", "global_step": 1968, "epoch": 0.07904566815278949, "loss": 0.0097, "grad_norm": 3.711296319961548, "learning_rate": 4.03939393939394e-06, "num_tokens": 4434286.0, "completions/mean_length": 33.125, "completions/min_length": 33.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.125, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9655854105949402, "rewards/meter/std": 0.0028627016581594944, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9655854105949402, "rewards/total_composite/std": 0.0028627016581594944, "reward": 0.9655854105949402, "reward_std": 0.0028626781422644854, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012964806519448757, "sampling/sampling_logp_difference/max": 0.5334515571594238, "sampling/importance_sampling_ratio/min": 0.5865768194198608, "sampling/importance_sampling_ratio/mean": 1.0038316249847412, "sampling/importance_sampling_ratio/max": 1.5242719650268555, "entropy": 0.09357678238302469, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/high_mean": 0.007575757801532745, "clip_ratio/high_max": 0.007575757801532745, "clip_ratio/region_mean": 0.011363636702299118, "reward_total_mean": 0.9655854105949402, "reward_meter_mean": 0.9655854105949402, "reward_meter_std": 0.0028627016581594944, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9655854105949402, "reward_total_composite_std": 0.0028627016581594944} {"timestamp_utc": "2026-04-12T01:11:36Z", "mode": "train", "global_step": 1969, "epoch": 0.07908583363457444, "loss": 0.0, "grad_norm": 0.08461344242095947, "learning_rate": 4.036363636363637e-06, "num_tokens": 4436078.0, "completions/mean_length": 63.0, "completions/min_length": 63.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.0, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9970186352729797, "rewards/meter/std": 3.89859178540064e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9970186352729797, "rewards/total_composite/std": 3.89859178540064e-06, "reward": 0.9970186352729797, "reward_std": 3.889565959980246e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.002396326744928956, "sampling/sampling_logp_difference/max": 0.5296306610107422, "sampling/importance_sampling_ratio/min": 0.5888224244117737, "sampling/importance_sampling_ratio/mean": 1.0005252361297607, "sampling/importance_sampling_ratio/max": 1.0470123291015625, "entropy": 0.012546915793791413, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9970186352729797, "reward_meter_mean": 0.9970186352729797, "reward_meter_std": 3.89859178540064e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9970186352729797, "reward_total_composite_std": 3.89859178540064e-06} {"timestamp_utc": "2026-04-12T01:11:41Z", "mode": "train", "global_step": 1970, "epoch": 0.0791259991163594, "loss": 0.0007, "grad_norm": 4.984862327575684, "learning_rate": 4.033333333333333e-06, "num_tokens": 4437774.0, "completions/mean_length": 33.0, "completions/min_length": 33.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9650729894638062, "rewards/meter/std": 0.0013392050750553608, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9650729894638062, "rewards/total_composite/std": 0.0013392050750553608, "reward": 0.9650729894638062, "reward_std": 0.0013392041437327862, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010843876749277115, "sampling/sampling_logp_difference/max": 0.6145038604736328, "sampling/importance_sampling_ratio/min": 0.5409092307090759, "sampling/importance_sampling_ratio/mean": 1.0015711784362793, "sampling/importance_sampling_ratio/max": 1.5502804517745972, "entropy": 0.08069896046072245, "clip_ratio/low_mean": 0.007575757801532745, "clip_ratio/low_min": 0.007575757801532745, "clip_ratio/high_mean": 0.0037878789007663727, "clip_ratio/high_max": 0.0037878789007663727, "clip_ratio/region_mean": 0.011363636702299118, "reward_total_mean": 0.9650729894638062, "reward_meter_mean": 0.9650729894638062, "reward_meter_std": 0.0013392050750553608, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9650729894638062, "reward_total_composite_std": 0.0013392050750553608} {"timestamp_utc": "2026-04-12T01:11:45Z", "mode": "train", "global_step": 1971, "epoch": 0.07916616459814435, "loss": 0.0063, "grad_norm": 2.97381854057312, "learning_rate": 4.030303030303031e-06, "num_tokens": 4439592.0, "completions/mean_length": 72.25, "completions/min_length": 70.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.25, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9976744651794434, "rewards/meter/std": 0.0007276128162629902, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9976744651794434, "rewards/total_composite/std": 0.0007276128162629902, "reward": 0.9976744651794434, "reward_std": 0.0007276106043718755, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03132731840014458, "sampling/sampling_logp_difference/max": 1.332169532775879, "sampling/importance_sampling_ratio/min": 0.2639040946960449, "sampling/importance_sampling_ratio/mean": 1.0109328031539917, "sampling/importance_sampling_ratio/max": 1.839429497718811, "entropy": 0.24298556335270405, "clip_ratio/low_mean": 0.008561643771827221, "clip_ratio/low_min": 0.008561643771827221, "clip_ratio/high_mean": 0.01056614622939378, "clip_ratio/high_max": 0.01056614622939378, "clip_ratio/region_mean": 0.019127790001221, "reward_total_mean": 0.9976744651794434, "reward_meter_mean": 0.9976744651794434, "reward_meter_std": 0.0007276128162629902, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9976744651794434, "reward_total_composite_std": 0.0007276128162629902} {"timestamp_utc": "2026-04-12T01:11:50Z", "mode": "train", "global_step": 1972, "epoch": 0.0792063300799293, "loss": -0.0272, "grad_norm": 5.1056904792785645, "learning_rate": 4.027272727272727e-06, "num_tokens": 4441521.0, "completions/mean_length": 75.125, "completions/min_length": 70.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.125, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9973230361938477, "rewards/meter/std": 0.0019267015159130096, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973230361938477, "rewards/total_composite/std": 0.0019267015159130096, "reward": 0.9973230361938477, "reward_std": 0.0019267020979896188, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04758862033486366, "sampling/sampling_logp_difference/max": 1.3674206733703613, "sampling/importance_sampling_ratio/min": 0.25476324558258057, "sampling/importance_sampling_ratio/mean": 1.0059734582901, "sampling/importance_sampling_ratio/max": 1.9101423025131226, "entropy": 0.47906696051359177, "clip_ratio/low_mean": 0.015499898232519627, "clip_ratio/low_min": 0.015499898232519627, "clip_ratio/high_mean": 0.02989264251664281, "clip_ratio/high_max": 0.02989264251664281, "clip_ratio/region_mean": 0.045392540749162436, "reward_total_mean": 0.9973230361938477, "reward_meter_mean": 0.9973230361938477, "reward_meter_std": 0.0019267015159130096, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9973230361938477, "reward_total_composite_std": 0.0019267015159130096} {"timestamp_utc": "2026-04-12T01:11:56Z", "mode": "train", "global_step": 1973, "epoch": 0.07924649556171426, "loss": -0.0034, "grad_norm": 1.4667316675186157, "learning_rate": 4.024242424242424e-06, "num_tokens": 4443935.0, "completions/mean_length": 134.75, "completions/min_length": 133.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 134.75, "completions/min_terminated_length": 133.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.8778820037841797, "rewards/meter/std": 0.01677864044904709, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7524703145027161, "rewards/total_composite/std": 0.014381703920662403, "reward": 0.7524703145027161, "reward_std": 0.014381689950823784, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009153633378446102, "sampling/sampling_logp_difference/max": 1.524185061454773, "sampling/importance_sampling_ratio/min": 0.21779848635196686, "sampling/importance_sampling_ratio/mean": 1.0035121440887451, "sampling/importance_sampling_ratio/max": 1.7678629159927368, "entropy": 0.0543739995919168, "clip_ratio/low_mean": 0.0028056252049282193, "clip_ratio/low_min": 0.0028056252049282193, "clip_ratio/high_mean": 0.003703703638166189, "clip_ratio/high_max": 0.003703703638166189, "clip_ratio/region_mean": 0.0065093288430944085, "reward_total_mean": 0.7524703145027161, "reward_meter_mean": 0.8778820037841797, "reward_meter_std": 0.01677864044904709, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7524703145027161, "reward_total_composite_std": 0.014381703920662403} {"timestamp_utc": "2026-04-12T01:12:01Z", "mode": "train", "global_step": 1974, "epoch": 0.07928666104349921, "loss": -0.0005, "grad_norm": 1.7984775304794312, "learning_rate": 4.0212121212121216e-06, "num_tokens": 4446120.0, "completions/mean_length": 100.125, "completions/min_length": 99.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.125, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.9985292553901672, "rewards/meter/std": 0.000194300344446674, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9985292553901672, "rewards/total_composite/std": 0.000194300344446674, "reward": 0.9985292553901672, "reward_std": 0.00019429647363722324, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01695932447910309, "sampling/sampling_logp_difference/max": 1.2210431098937988, "sampling/importance_sampling_ratio/min": 0.2949223816394806, "sampling/importance_sampling_ratio/mean": 1.0018852949142456, "sampling/importance_sampling_ratio/max": 1.6485373973846436, "entropy": 0.10710672754794359, "clip_ratio/low_mean": 0.007525252411141992, "clip_ratio/low_min": 0.007525252411141992, "clip_ratio/high_mean": 0.006263126386329532, "clip_ratio/high_max": 0.006263126386329532, "clip_ratio/region_mean": 0.013788378797471523, "reward_total_mean": 0.9985292553901672, "reward_meter_mean": 0.9985292553901672, "reward_meter_std": 0.000194300344446674, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9985292553901672, "reward_total_composite_std": 0.000194300344446674} {"timestamp_utc": "2026-04-12T01:12:06Z", "mode": "train", "global_step": 1975, "epoch": 0.07932682652528417, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.018181818181818e-06, "num_tokens": 4447928.0, "completions/mean_length": 63.0, "completions/min_length": 63.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.0, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9970200061798096, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9970200061798096, "rewards/total_composite/std": 0.0, "reward": 0.9970200061798096, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0013655778020620346, "sampling/sampling_logp_difference/max": 0.04307138919830322, "sampling/importance_sampling_ratio/min": 0.9628649353981018, "sampling/importance_sampling_ratio/mean": 1.0012253522872925, "sampling/importance_sampling_ratio/max": 1.044012427330017, "entropy": 0.011537018930539489, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9970200061798096, "reward_meter_mean": 0.9970200061798096, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9970200061798096, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:12:12Z", "mode": "train", "global_step": 1976, "epoch": 0.07936699200706912, "loss": 0.0074, "grad_norm": 3.668858766555786, "learning_rate": 4.015151515151515e-06, "num_tokens": 4451433.0, "completions/mean_length": 222.125, "completions/min_length": 216.0, "completions/max_length": 229.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 222.125, "completions/min_terminated_length": 216.0, "completions/max_terminated_length": 229.0, "rewards/meter/mean": 0.9412789940834045, "rewards/meter/std": 0.15677009522914886, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9695513248443604, "rewards/repeat_penalty/std": 0.042069755494594574, "rewards/total_composite/mean": 0.8963793516159058, "rewards/total_composite/std": 0.17684437334537506, "reward": 0.8963793516159058, "reward_std": 0.17684435844421387, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04191061109304428, "sampling/sampling_logp_difference/max": 5.00631856918335, "sampling/importance_sampling_ratio/min": 0.0066955070942640305, "sampling/importance_sampling_ratio/mean": 0.9989370107650757, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.18782638758420944, "clip_ratio/low_mean": 0.0077665450517088175, "clip_ratio/low_min": 0.0077665450517088175, "clip_ratio/high_mean": 0.022601958131417632, "clip_ratio/high_max": 0.022601958131417632, "clip_ratio/region_mean": 0.03036850318312645, "reward_total_mean": 0.8963793516159058, "reward_meter_mean": 0.9412789940834045, "reward_meter_std": 0.15677009522914886, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9695513248443604, "reward_repeat_penalty_std": 0.042069755494594574, "reward_total_composite_mean": 0.8963793516159058, "reward_total_composite_std": 0.17684437334537506} {"timestamp_utc": "2026-04-12T01:12:19Z", "mode": "train", "global_step": 1977, "epoch": 0.07940715748885407, "loss": 0.0098, "grad_norm": 2.525111436843872, "learning_rate": 4.0121212121212125e-06, "num_tokens": 4454371.0, "completions/mean_length": 184.25, "completions/min_length": 179.0, "completions/max_length": 189.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 184.25, "completions/min_terminated_length": 179.0, "completions/max_terminated_length": 189.0, "rewards/meter/mean": 0.993905782699585, "rewards/meter/std": 0.00516868568956852, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.930555522441864, "rewards/repeat_penalty/std": 0.08266931027173996, "rewards/total_composite/mean": 0.9247466325759888, "rewards/total_composite/std": 0.08044659346342087, "reward": 0.9247466325759888, "reward_std": 0.08044657856225967, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.049122218042612076, "sampling/sampling_logp_difference/max": 1.7033472061157227, "sampling/importance_sampling_ratio/min": 0.18207307159900665, "sampling/importance_sampling_ratio/mean": 1.0052932500839233, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37936991080641747, "clip_ratio/low_mean": 0.011562932399101555, "clip_ratio/low_min": 0.011562932399101555, "clip_ratio/high_mean": 0.02449705172330141, "clip_ratio/high_max": 0.02449705172330141, "clip_ratio/region_mean": 0.036059984122402966, "reward_total_mean": 0.9247466325759888, "reward_meter_mean": 0.993905782699585, "reward_meter_std": 0.00516868568956852, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.930555522441864, "reward_repeat_penalty_std": 0.08266931027173996, "reward_total_composite_mean": 0.9247466325759888, "reward_total_composite_std": 0.08044659346342087} {"timestamp_utc": "2026-04-12T01:12:23Z", "mode": "train", "global_step": 1978, "epoch": 0.07944732297063903, "loss": 0.0005, "grad_norm": 5.933767318725586, "learning_rate": 4.009090909090909e-06, "num_tokens": 4456189.0, "completions/mean_length": 76.25, "completions/min_length": 74.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.25, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.944914698600769, "rewards/meter/std": 0.14462126791477203, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.944914698600769, "rewards/total_composite/std": 0.14462126791477203, "reward": 0.944914698600769, "reward_std": 0.14462128281593323, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0594903789460659, "sampling/sampling_logp_difference/max": 1.3728089332580566, "sampling/importance_sampling_ratio/min": 0.2533941864967346, "sampling/importance_sampling_ratio/mean": 1.005175232887268, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4438166059553623, "clip_ratio/low_mean": 0.009999999776482582, "clip_ratio/low_min": 0.009999999776482582, "clip_ratio/high_mean": 0.03106398310046643, "clip_ratio/high_max": 0.03106398310046643, "clip_ratio/region_mean": 0.04106398287694901, "reward_total_mean": 0.944914698600769, "reward_meter_mean": 0.944914698600769, "reward_meter_std": 0.14462126791477203, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.944914698600769, "reward_total_composite_std": 0.14462126791477203} {"timestamp_utc": "2026-04-12T01:12:28Z", "mode": "train", "global_step": 1979, "epoch": 0.07948748845242398, "loss": 0.0023, "grad_norm": 1.3233561515808105, "learning_rate": 4.006060606060607e-06, "num_tokens": 4457998.0, "completions/mean_length": 66.125, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.125, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9989874362945557, "rewards/meter/std": 5.0926621042890474e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989874362945557, "rewards/total_composite/std": 5.0926621042890474e-05, "reward": 0.9989874362945557, "reward_std": 5.0924856623169035e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010206947103142738, "sampling/sampling_logp_difference/max": 0.8685154914855957, "sampling/importance_sampling_ratio/min": 0.41957396268844604, "sampling/importance_sampling_ratio/mean": 0.9985904693603516, "sampling/importance_sampling_ratio/max": 1.4895697832107544, "entropy": 0.03965302975848317, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/high_mean": 0.0037878789007663727, "clip_ratio/high_max": 0.0037878789007663727, "clip_ratio/region_mean": 0.0056535504991188645, "reward_total_mean": 0.9989874362945557, "reward_meter_mean": 0.9989874362945557, "reward_meter_std": 5.0926621042890474e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9989874362945557, "reward_total_composite_std": 5.0926621042890474e-05} {"timestamp_utc": "2026-04-12T01:12:35Z", "mode": "train", "global_step": 1980, "epoch": 0.07952765393420894, "loss": 0.087, "grad_norm": 2.8739843368530273, "learning_rate": 4.003030303030303e-06, "num_tokens": 4461140.0, "completions/mean_length": 199.75, "completions/min_length": 169.0, "completions/max_length": 222.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 199.75, "completions/min_terminated_length": 169.0, "completions/max_terminated_length": 222.0, "rewards/meter/mean": 0.9881318211555481, "rewards/meter/std": 0.012296332977712154, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8166666626930237, "rewards/repeat_penalty/std": 0.05781134217977524, "rewards/total_composite/mean": 0.7281157374382019, "rewards/total_composite/std": 0.11554723232984543, "reward": 0.7281157374382019, "reward_std": 0.11554723978042603, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031819190829992294, "sampling/sampling_logp_difference/max": 4.230698585510254, "sampling/importance_sampling_ratio/min": 0.014542227610945702, "sampling/importance_sampling_ratio/mean": 0.9996204972267151, "sampling/importance_sampling_ratio/max": 1.922302007675171, "entropy": 0.16436910070478916, "clip_ratio/low_mean": 0.010806537233293056, "clip_ratio/low_min": 0.010806537233293056, "clip_ratio/high_mean": 0.013719010865315795, "clip_ratio/high_max": 0.013719010865315795, "clip_ratio/region_mean": 0.02452554809860885, "reward_total_mean": 0.7281157374382019, "reward_meter_mean": 0.9881318211555481, "reward_meter_std": 0.012296332977712154, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8166666626930237, "reward_repeat_penalty_std": 0.05781134217977524, "reward_total_composite_mean": 0.7281157374382019, "reward_total_composite_std": 0.11554723232984543} {"timestamp_utc": "2026-04-12T01:12:40Z", "mode": "train", "global_step": 1981, "epoch": 0.07956781941599389, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.000000000000001e-06, "num_tokens": 4463220.0, "completions/mean_length": 90.0, "completions/min_length": 90.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.0, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9955950975418091, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9955950975418091, "rewards/total_composite/std": 0.0, "reward": 0.9955950975418091, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0013981742085888982, "sampling/sampling_logp_difference/max": 0.12119700014591217, "sampling/importance_sampling_ratio/min": 0.937319278717041, "sampling/importance_sampling_ratio/mean": 1.0010671615600586, "sampling/importance_sampling_ratio/max": 1.1288472414016724, "entropy": 0.011815825360827148, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9955950975418091, "reward_meter_mean": 0.9955950975418091, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9955950975418091, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:12:45Z", "mode": "train", "global_step": 1982, "epoch": 0.07960798489777884, "loss": 0.0058, "grad_norm": 6.157790660858154, "learning_rate": 3.996969696969698e-06, "num_tokens": 4465131.0, "completions/mean_length": 76.875, "completions/min_length": 73.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.875, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9915469884872437, "rewards/meter/std": 0.0046110316179692745, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9915469884872437, "rewards/total_composite/std": 0.0046110316179692745, "reward": 0.9915469884872437, "reward_std": 0.004611044656485319, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03136689215898514, "sampling/sampling_logp_difference/max": 1.0577783584594727, "sampling/importance_sampling_ratio/min": 0.3472263514995575, "sampling/importance_sampling_ratio/mean": 1.0080691576004028, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.357822060585022, "clip_ratio/low_mean": 0.008034686907194555, "clip_ratio/low_min": 0.008034686907194555, "clip_ratio/high_mean": 0.01946915965527296, "clip_ratio/high_max": 0.01946915965527296, "clip_ratio/region_mean": 0.027503846562467515, "reward_total_mean": 0.9915469884872437, "reward_meter_mean": 0.9915469884872437, "reward_meter_std": 0.0046110316179692745, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9915469884872437, "reward_total_composite_std": 0.0046110316179692745} {"timestamp_utc": "2026-04-12T01:12:49Z", "mode": "train", "global_step": 1983, "epoch": 0.0796481503795638, "loss": -0.0135, "grad_norm": 4.942656993865967, "learning_rate": 3.993939393939394e-06, "num_tokens": 4466706.0, "completions/mean_length": 32.875, "completions/min_length": 32.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.875, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9034555554389954, "rewards/meter/std": 0.1775951385498047, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9034555554389954, "rewards/total_composite/std": 0.1775951385498047, "reward": 0.9034555554389954, "reward_std": 0.1775951385498047, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0072293514385819435, "sampling/sampling_logp_difference/max": 0.4028921127319336, "sampling/importance_sampling_ratio/min": 0.6683841943740845, "sampling/importance_sampling_ratio/mean": 1.0033267736434937, "sampling/importance_sampling_ratio/max": 1.1638069152832031, "entropy": 0.06188688660040498, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.00390625, "reward_total_mean": 0.9034555554389954, "reward_meter_mean": 0.9034555554389954, "reward_meter_std": 0.1775951385498047, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9034555554389954, "reward_total_composite_std": 0.1775951385498047} {"timestamp_utc": "2026-04-12T01:12:54Z", "mode": "train", "global_step": 1984, "epoch": 0.07968831586134875, "loss": -0.0081, "grad_norm": 0.42240381240844727, "learning_rate": 3.990909090909092e-06, "num_tokens": 4468579.0, "completions/mean_length": 67.125, "completions/min_length": 67.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.125, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9162052869796753, "rewards/meter/std": 0.0074835000559687614, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9162052869796753, "rewards/total_composite/std": 0.0074835000559687614, "reward": 0.9162052869796753, "reward_std": 0.007483497262001038, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004512776155024767, "sampling/sampling_logp_difference/max": 0.38767099380493164, "sampling/importance_sampling_ratio/min": 0.7210829854011536, "sampling/importance_sampling_ratio/mean": 1.0025385618209839, "sampling/importance_sampling_ratio/max": 1.473544955253601, "entropy": 0.03432013071142137, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/high_mean": 0.0018382353009656072, "clip_ratio/high_max": 0.0018382353009656072, "clip_ratio/region_mean": 0.003703906899318099, "reward_total_mean": 0.9162052869796753, "reward_meter_mean": 0.9162052869796753, "reward_meter_std": 0.0074835000559687614, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9162052869796753, "reward_total_composite_std": 0.0074835000559687614} {"timestamp_utc": "2026-04-12T01:12:59Z", "mode": "train", "global_step": 1985, "epoch": 0.0797284813431337, "loss": 0.0002, "grad_norm": 1.1522923707962036, "learning_rate": 3.987878787878788e-06, "num_tokens": 4470379.0, "completions/mean_length": 58.0, "completions/min_length": 58.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.994735836982727, "rewards/meter/std": 6.81514575262554e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.994735836982727, "rewards/total_composite/std": 6.81514575262554e-05, "reward": 0.994735836982727, "reward_std": 6.814543303335086e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0007778764702379704, "sampling/sampling_logp_difference/max": 0.019494354724884033, "sampling/importance_sampling_ratio/min": 0.9891203045845032, "sampling/importance_sampling_ratio/mean": 1.0006917715072632, "sampling/importance_sampling_ratio/max": 1.0196856260299683, "entropy": 0.007963738986290991, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.994735836982727, "reward_meter_mean": 0.994735836982727, "reward_meter_std": 6.81514575262554e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.994735836982727, "reward_total_composite_std": 6.81514575262554e-05} {"timestamp_utc": "2026-04-12T01:13:03Z", "mode": "train", "global_step": 1986, "epoch": 0.07976864682491866, "loss": 0.0087, "grad_norm": 13.343822479248047, "learning_rate": 3.984848484848485e-06, "num_tokens": 4471880.0, "completions/mean_length": 40.625, "completions/min_length": 37.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.625, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.5094391703605652, "rewards/meter/std": 0.4160431921482086, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.47001466155052185, "rewards/total_composite/std": 0.39801424741744995, "reward": 0.47001466155052185, "reward_std": 0.39801424741744995, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10109325498342514, "sampling/sampling_logp_difference/max": 1.5112323760986328, "sampling/importance_sampling_ratio/min": 0.22063790261745453, "sampling/importance_sampling_ratio/mean": 1.0084450244903564, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6491754315793514, "clip_ratio/low_mean": 0.05110658518970013, "clip_ratio/low_min": 0.05110658518970013, "clip_ratio/high_mean": 0.03596452111378312, "clip_ratio/high_max": 0.03596452111378312, "clip_ratio/region_mean": 0.08707110630348325, "reward_total_mean": 0.47001466155052185, "reward_meter_mean": 0.5094391703605652, "reward_meter_std": 0.4160431921482086, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.47001466155052185, "reward_total_composite_std": 0.39801424741744995} {"timestamp_utc": "2026-04-12T01:13:09Z", "mode": "train", "global_step": 1987, "epoch": 0.07980881230670361, "loss": -0.0103, "grad_norm": 4.379970073699951, "learning_rate": 3.9818181818181825e-06, "num_tokens": 4474379.0, "completions/mean_length": 149.375, "completions/min_length": 137.0, "completions/max_length": 156.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 149.375, "completions/min_terminated_length": 137.0, "completions/max_terminated_length": 156.0, "rewards/meter/mean": 0.9875733852386475, "rewards/meter/std": 0.02084759995341301, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9405426383018494, "rewards/total_composite/std": 0.10836569964885712, "reward": 0.9405426383018494, "reward_std": 0.10836568474769592, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04869993403553963, "sampling/sampling_logp_difference/max": 1.6888227462768555, "sampling/importance_sampling_ratio/min": 0.18473687767982483, "sampling/importance_sampling_ratio/mean": 1.0045157670974731, "sampling/importance_sampling_ratio/max": 1.912219524383545, "entropy": 0.4626317396759987, "clip_ratio/low_mean": 0.010287561686709523, "clip_ratio/low_min": 0.010287561686709523, "clip_ratio/high_mean": 0.02709868340753019, "clip_ratio/high_max": 0.02709868340753019, "clip_ratio/region_mean": 0.03738624509423971, "reward_total_mean": 0.9405426383018494, "reward_meter_mean": 0.9875733852386475, "reward_meter_std": 0.02084759995341301, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.9405426383018494, "reward_total_composite_std": 0.10836569964885712} {"timestamp_utc": "2026-04-12T01:13:14Z", "mode": "train", "global_step": 1988, "epoch": 0.07984897778848857, "loss": 0.0169, "grad_norm": 6.39749002456665, "learning_rate": 3.978787878787879e-06, "num_tokens": 4476330.0, "completions/mean_length": 76.875, "completions/min_length": 75.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.875, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.988801121711731, "rewards/meter/std": 0.014979206025600433, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.988801121711731, "rewards/total_composite/std": 0.014979206025600433, "reward": 0.988801121711731, "reward_std": 0.014979207888245583, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05862836539745331, "sampling/sampling_logp_difference/max": 1.5416898727416992, "sampling/importance_sampling_ratio/min": 0.21401913464069366, "sampling/importance_sampling_ratio/mean": 1.0134505033493042, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4746681675314903, "clip_ratio/low_mean": 0.008121267077513039, "clip_ratio/low_min": 0.008121267077513039, "clip_ratio/high_mean": 0.022676826687529683, "clip_ratio/high_max": 0.022676826687529683, "clip_ratio/region_mean": 0.030798093765042722, "reward_total_mean": 0.988801121711731, "reward_meter_mean": 0.988801121711731, "reward_meter_std": 0.014979206025600433, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.988801121711731, "reward_total_composite_std": 0.014979206025600433} {"timestamp_utc": "2026-04-12T01:13:18Z", "mode": "train", "global_step": 1989, "epoch": 0.07988914327027352, "loss": -0.0063, "grad_norm": 5.730335712432861, "learning_rate": 3.975757575757576e-06, "num_tokens": 4478142.0, "completions/mean_length": 67.5, "completions/min_length": 67.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.5, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9175406694412231, "rewards/meter/std": 0.021055592224001884, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9175406694412231, "rewards/total_composite/std": 0.021055592224001884, "reward": 0.9175406694412231, "reward_std": 0.021055590361356735, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014552988111972809, "sampling/sampling_logp_difference/max": 0.96074378490448, "sampling/importance_sampling_ratio/min": 0.38260820508003235, "sampling/importance_sampling_ratio/mean": 1.0017634630203247, "sampling/importance_sampling_ratio/max": 1.4384418725967407, "entropy": 0.08089068345725536, "clip_ratio/low_mean": 0.005597014795057476, "clip_ratio/low_min": 0.005597014795057476, "clip_ratio/high_mean": 0.0036764706019312143, "clip_ratio/high_max": 0.0036764706019312143, "clip_ratio/region_mean": 0.00927348539698869, "reward_total_mean": 0.9175406694412231, "reward_meter_mean": 0.9175406694412231, "reward_meter_std": 0.021055592224001884, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9175406694412231, "reward_total_composite_std": 0.021055592224001884} {"timestamp_utc": "2026-04-12T01:13:25Z", "mode": "train", "global_step": 1990, "epoch": 0.07992930875205848, "loss": -0.0004, "grad_norm": 2.3331921100616455, "learning_rate": 3.972727272727273e-06, "num_tokens": 4481575.0, "completions/mean_length": 216.125, "completions/min_length": 206.0, "completions/max_length": 222.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 216.125, "completions/min_terminated_length": 206.0, "completions/max_terminated_length": 222.0, "rewards/meter/mean": 0.998195230960846, "rewards/meter/std": 0.0007794296252541244, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9043561220169067, "rewards/repeat_penalty/std": 0.053097911179065704, "rewards/total_composite/mean": 0.9027365446090698, "rewards/total_composite/std": 0.053245700895786285, "reward": 0.9027365446090698, "reward_std": 0.05324570834636688, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023263096809387207, "sampling/sampling_logp_difference/max": 1.5214588642120361, "sampling/importance_sampling_ratio/min": 0.21839307248592377, "sampling/importance_sampling_ratio/mean": 1.0023928880691528, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09882919304072857, "clip_ratio/low_mean": 0.005168477655388415, "clip_ratio/low_min": 0.005168477655388415, "clip_ratio/high_mean": 0.016820423072203994, "clip_ratio/high_max": 0.016820423072203994, "clip_ratio/region_mean": 0.02198890072759241, "reward_total_mean": 0.9027365446090698, "reward_meter_mean": 0.998195230960846, "reward_meter_std": 0.0007794296252541244, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9043561220169067, "reward_repeat_penalty_std": 0.053097911179065704, "reward_total_composite_mean": 0.9027365446090698, "reward_total_composite_std": 0.053245700895786285} {"timestamp_utc": "2026-04-12T01:13:33Z", "mode": "train", "global_step": 1991, "epoch": 0.07996947423384343, "loss": 0.0014, "grad_norm": 1.446068286895752, "learning_rate": 3.96969696969697e-06, "num_tokens": 4485554.0, "completions/mean_length": 288.375, "completions/min_length": 287.0, "completions/max_length": 290.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 288.375, "completions/min_terminated_length": 287.0, "completions/max_terminated_length": 290.0, "rewards/meter/mean": 0.9093496203422546, "rewards/meter/std": 0.01770671270787716, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6796875, "rewards/repeat_penalty/std": 0.022097086533904076, "rewards/total_composite/mean": 0.411979079246521, "rewards/total_composite/std": 0.013528194278478622, "reward": 0.411979079246521, "reward_std": 0.013528193347156048, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02342934161424637, "sampling/sampling_logp_difference/max": 13.880716323852539, "sampling/importance_sampling_ratio/min": 9.368746418658702e-07, "sampling/importance_sampling_ratio/mean": 0.9994128942489624, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.059712667018175125, "clip_ratio/low_mean": 0.001727131224470213, "clip_ratio/low_min": 0.001727131224470213, "clip_ratio/high_mean": 0.007373977248789743, "clip_ratio/high_max": 0.007373977248789743, "clip_ratio/region_mean": 0.009101108473259956, "reward_total_mean": 0.411979079246521, "reward_meter_mean": 0.9093496203422546, "reward_meter_std": 0.01770671270787716, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6796875, "reward_repeat_penalty_std": 0.022097086533904076, "reward_total_composite_mean": 0.411979079246521, "reward_total_composite_std": 0.013528194278478622} {"timestamp_utc": "2026-04-12T01:13:38Z", "mode": "train", "global_step": 1992, "epoch": 0.08000963971562838, "loss": -0.0152, "grad_norm": 1.818966031074524, "learning_rate": 3.966666666666667e-06, "num_tokens": 4488257.0, "completions/mean_length": 157.875, "completions/min_length": 150.0, "completions/max_length": 159.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 157.875, "completions/min_terminated_length": 150.0, "completions/max_terminated_length": 159.0, "rewards/meter/mean": 0.9970660209655762, "rewards/meter/std": 0.00133254355750978, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7777777910232544, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7754957675933838, "rewards/total_composite/std": 0.0010364169720560312, "reward": 0.7754957675933838, "reward_std": 0.001036411034874618, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.002477371832355857, "sampling/sampling_logp_difference/max": 0.8110690116882324, "sampling/importance_sampling_ratio/min": 0.44438278675079346, "sampling/importance_sampling_ratio/mean": 1.0000977516174316, "sampling/importance_sampling_ratio/max": 1.1387183666229248, "entropy": 0.011172075115609914, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7754957675933838, "reward_meter_mean": 0.9970660209655762, "reward_meter_std": 0.00133254355750978, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7777777910232544, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7754957675933838, "reward_total_composite_std": 0.0010364169720560312} {"timestamp_utc": "2026-04-12T01:13:43Z", "mode": "train", "global_step": 1993, "epoch": 0.08004980519741334, "loss": -0.0053, "grad_norm": 3.636791467666626, "learning_rate": 3.963636363636364e-06, "num_tokens": 4490233.0, "completions/mean_length": 73.0, "completions/min_length": 72.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.0, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9986278414726257, "rewards/meter/std": 0.0007868955726735294, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9986278414726257, "rewards/total_composite/std": 0.0007868955726735294, "reward": 0.9986278414726257, "reward_std": 0.0007868955144658685, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03113717958331108, "sampling/sampling_logp_difference/max": 1.0237507820129395, "sampling/importance_sampling_ratio/min": 0.3592449724674225, "sampling/importance_sampling_ratio/mean": 1.0071430206298828, "sampling/importance_sampling_ratio/max": 1.5565049648284912, "entropy": 0.21745091304183006, "clip_ratio/low_mean": 0.0034722222480922937, "clip_ratio/low_min": 0.0034722222480922937, "clip_ratio/high_mean": 0.011917525203898549, "clip_ratio/high_max": 0.011917525203898549, "clip_ratio/region_mean": 0.015389747451990843, "reward_total_mean": 0.9986278414726257, "reward_meter_mean": 0.9986278414726257, "reward_meter_std": 0.0007868955726735294, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9986278414726257, "reward_total_composite_std": 0.0007868955726735294} {"timestamp_utc": "2026-04-12T01:13:49Z", "mode": "train", "global_step": 1994, "epoch": 0.08008997067919829, "loss": -0.0042, "grad_norm": 3.498656749725342, "learning_rate": 3.960606060606061e-06, "num_tokens": 4492892.0, "completions/mean_length": 147.375, "completions/min_length": 144.0, "completions/max_length": 152.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 147.375, "completions/min_terminated_length": 144.0, "completions/max_terminated_length": 152.0, "rewards/meter/mean": 0.9948342442512512, "rewards/meter/std": 0.006030126474797726, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9948342442512512, "rewards/total_composite/std": 0.006030126474797726, "reward": 0.9948342442512512, "reward_std": 0.006030108779668808, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06232795864343643, "sampling/sampling_logp_difference/max": 1.819002628326416, "sampling/importance_sampling_ratio/min": 0.16218742728233337, "sampling/importance_sampling_ratio/mean": 1.0064527988433838, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4670833945274353, "clip_ratio/low_mean": 0.006076388992369175, "clip_ratio/low_min": 0.006076388992369175, "clip_ratio/high_mean": 0.03634089510887861, "clip_ratio/high_max": 0.03634089510887861, "clip_ratio/region_mean": 0.04241728410124779, "reward_total_mean": 0.9948342442512512, "reward_meter_mean": 0.9948342442512512, "reward_meter_std": 0.006030126474797726, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9948342442512512, "reward_total_composite_std": 0.006030126474797726} {"timestamp_utc": "2026-04-12T01:13:53Z", "mode": "train", "global_step": 1995, "epoch": 0.08013013616098325, "loss": 0.0012, "grad_norm": 0.7663399577140808, "learning_rate": 3.957575757575758e-06, "num_tokens": 4494276.0, "completions/mean_length": 31.0, "completions/min_length": 31.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.9957044124603271, "rewards/meter/std": 3.0345732739078812e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957044124603271, "rewards/total_composite/std": 3.0345732739078812e-05, "reward": 0.9957044124603271, "reward_std": 3.034573092008941e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02179253287613392, "sampling/sampling_logp_difference/max": 4.0259270668029785, "sampling/importance_sampling_ratio/min": 0.01784687116742134, "sampling/importance_sampling_ratio/mean": 0.9947730898857117, "sampling/importance_sampling_ratio/max": 1.0392725467681885, "entropy": 0.03334390290547162, "clip_ratio/low_mean": 0.012096773833036423, "clip_ratio/low_min": 0.012096773833036423, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.012096773833036423, "reward_total_mean": 0.9957044124603271, "reward_meter_mean": 0.9957044124603271, "reward_meter_std": 3.0345732739078812e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9957044124603271, "reward_total_composite_std": 3.0345732739078812e-05} {"timestamp_utc": "2026-04-12T01:13:57Z", "mode": "train", "global_step": 1996, "epoch": 0.0801703016427682, "loss": -0.0084, "grad_norm": 11.98099422454834, "learning_rate": 3.954545454545454e-06, "num_tokens": 4495750.0, "completions/mean_length": 34.25, "completions/min_length": 34.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.25, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9950810074806213, "rewards/meter/std": 0.0017581552965566516, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9950810074806213, "rewards/total_composite/std": 0.0017581552965566516, "reward": 0.9950810074806213, "reward_std": 0.0017581689171493053, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015304782427847385, "sampling/sampling_logp_difference/max": 1.0786250829696655, "sampling/importance_sampling_ratio/min": 0.549746036529541, "sampling/importance_sampling_ratio/mean": 1.008811354637146, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.046473859809339046, "clip_ratio/low_mean": 0.011029411805793643, "clip_ratio/low_min": 0.011029411805793643, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.011029411805793643, "reward_total_mean": 0.9950810074806213, "reward_meter_mean": 0.9950810074806213, "reward_meter_std": 0.0017581552965566516, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9950810074806213, "reward_total_composite_std": 0.0017581552965566516} {"timestamp_utc": "2026-04-12T01:14:02Z", "mode": "train", "global_step": 1997, "epoch": 0.08021046712455315, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.951515151515152e-06, "num_tokens": 4497582.0, "completions/mean_length": 69.0, "completions/min_length": 69.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.0, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9355690479278564, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9355690479278564, "rewards/total_composite/std": 0.0, "reward": 0.9355690479278564, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0048114219680428505, "sampling/sampling_logp_difference/max": 0.08930085599422455, "sampling/importance_sampling_ratio/min": 0.9659483432769775, "sampling/importance_sampling_ratio/mean": 1.004404067993164, "sampling/importance_sampling_ratio/max": 1.093409538269043, "entropy": 0.048666974529623985, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9355690479278564, "reward_meter_mean": 0.9355690479278564, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9355690479278564, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:14:06Z", "mode": "train", "global_step": 1998, "epoch": 0.08025063260633811, "loss": 0.0006, "grad_norm": 1.003632664680481, "learning_rate": 3.948484848484849e-06, "num_tokens": 4499102.0, "completions/mean_length": 30.0, "completions/min_length": 30.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.0, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.9927566647529602, "rewards/meter/std": 3.355137232574634e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9927566647529602, "rewards/total_composite/std": 3.355137232574634e-05, "reward": 0.9927566647529602, "reward_std": 3.355137232574634e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010052908211946487, "sampling/sampling_logp_difference/max": 0.4464629590511322, "sampling/importance_sampling_ratio/min": 0.6398874521255493, "sampling/importance_sampling_ratio/mean": 1.0059435367584229, "sampling/importance_sampling_ratio/max": 1.4227073192596436, "entropy": 0.05545296333730221, "clip_ratio/low_mean": 0.01666666753590107, "clip_ratio/low_min": 0.01666666753590107, "clip_ratio/high_mean": 0.004166666883975267, "clip_ratio/high_max": 0.004166666883975267, "clip_ratio/region_mean": 0.020833334419876337, "reward_total_mean": 0.9927566647529602, "reward_meter_mean": 0.9927566647529602, "reward_meter_std": 3.355137232574634e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9927566647529602, "reward_total_composite_std": 3.355137232574634e-05} {"timestamp_utc": "2026-04-12T01:14:11Z", "mode": "train", "global_step": 1999, "epoch": 0.08029079808812306, "loss": -0.0049, "grad_norm": 2.3003454208374023, "learning_rate": 3.945454545454545e-06, "num_tokens": 4501021.0, "completions/mean_length": 73.875, "completions/min_length": 71.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.875, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9951367378234863, "rewards/meter/std": 0.003929613158106804, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9951367378234863, "rewards/total_composite/std": 0.003929613158106804, "reward": 0.9951367378234863, "reward_std": 0.003929623868316412, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04188167303800583, "sampling/sampling_logp_difference/max": 1.4766731262207031, "sampling/importance_sampling_ratio/min": 0.22839628159999847, "sampling/importance_sampling_ratio/mean": 1.0134656429290771, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40391664765775204, "clip_ratio/low_mean": 0.008561643771827221, "clip_ratio/low_min": 0.008561643771827221, "clip_ratio/high_mean": 0.02339910331647843, "clip_ratio/high_max": 0.02339910331647843, "clip_ratio/region_mean": 0.03196074708830565, "reward_total_mean": 0.9951367378234863, "reward_meter_mean": 0.9951367378234863, "reward_meter_std": 0.003929613158106804, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9951367378234863, "reward_total_composite_std": 0.003929613158106804} {"timestamp_utc": "2026-04-12T01:14:17Z", "mode": "train", "global_step": 2000, "epoch": 0.08033096356990801, "loss": 0.0069, "grad_norm": 3.8840928077697754, "learning_rate": 3.942424242424243e-06, "num_tokens": 4503974.0, "completions/mean_length": 147.125, "completions/min_length": 141.0, "completions/max_length": 153.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 147.125, "completions/min_terminated_length": 141.0, "completions/max_terminated_length": 153.0, "rewards/meter/mean": 0.9887276887893677, "rewards/meter/std": 0.01330604124814272, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9359963536262512, "rewards/total_composite/std": 0.07730759680271149, "reward": 0.9359963536262512, "reward_std": 0.07730759680271149, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05714353546500206, "sampling/sampling_logp_difference/max": 2.683351516723633, "sampling/importance_sampling_ratio/min": 0.06833374500274658, "sampling/importance_sampling_ratio/mean": 1.0066624879837036, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3696784619241953, "clip_ratio/low_mean": 0.005059222981799394, "clip_ratio/low_min": 0.005059222981799394, "clip_ratio/high_mean": 0.026401946786791086, "clip_ratio/high_max": 0.026401946786791086, "clip_ratio/region_mean": 0.03146116976859048, "reward_total_mean": 0.9359963536262512, "reward_meter_mean": 0.9887276887893677, "reward_meter_std": 0.01330604124814272, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.9359963536262512, "reward_total_composite_std": 0.07730759680271149} {"timestamp_utc": "2026-04-12T01:15:25Z", "mode": "eval", "global_step": 2000, "epoch": 0.08033096356990801, "eval_loss": NaN, "eval_runtime": 67.8915, "eval_samples_per_second": 1.532, "eval_steps_per_second": 0.191, "eval_num_tokens": 4503974.0, "eval_completions/mean_length": 205.2596153846154, "eval_completions/min_length": 62.92307692307692, "eval_completions/max_length": 358.15384615384613, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 205.2596153846154, "eval_completions/min_terminated_length": 62.92307692307692, "eval_completions/max_terminated_length": 358.15384615384613, "eval_rewards/meter/mean": 0.7367691305967478, "eval_rewards/meter/std": 0.3722034446322001, "eval_rewards/count_adherence/mean": 0.9229050003565274, "eval_rewards/count_adherence/std": 0.11920174153951499, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/repeat_penalty/mean": 0.8324548143606919, "eval_rewards/repeat_penalty/std": 0.1328287388269718, "eval_rewards/total_composite/mean": 0.568704937513058, "eval_rewards/total_composite/std": 0.34218758459274584, "eval_reward": 0.568704937513058, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.017686504655732557, "eval_sampling/sampling_logp_difference/max": 1.1488754015702467, "eval_sampling/importance_sampling_ratio/min": 0.33040788540473354, "eval_sampling/importance_sampling_ratio/mean": 1.0038928618797889, "eval_sampling/importance_sampling_ratio/max": 1.4441173993624175, "eval_entropy": 0.16280974046542093, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.568704937513058, "eval_reward_meter_mean": 0.7367691305967478, "eval_reward_meter_std": 0.3722034446322001, "eval_reward_count_adherence_mean": 0.9229050003565274, "eval_reward_count_adherence_std": 0.11920174153951499, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_repeat_penalty_mean": 0.8324548143606919, "eval_reward_repeat_penalty_std": 0.1328287388269718, "eval_reward_total_composite_mean": 0.568704937513058, "eval_reward_total_composite_std": 0.34218758459274584} {"timestamp_utc": "2026-04-12T01:15:33Z", "mode": "train", "global_step": 2001, "epoch": 0.08037112905169297, "loss": 0.0013, "grad_norm": 1.4718478918075562, "learning_rate": 3.93939393939394e-06, "num_tokens": 4506302.0, "completions/mean_length": 122.0, "completions/min_length": 122.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.0, "completions/min_terminated_length": 122.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.9962645173072815, "rewards/meter/std": 4.681191057898104e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.8717328310012817, "rewards/total_composite/std": 0.050349388271570206, "reward": 0.8717328310012817, "reward_std": 0.05034938082098961, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007403408642858267, "sampling/sampling_logp_difference/max": 1.1984543800354004, "sampling/importance_sampling_ratio/min": 0.30166009068489075, "sampling/importance_sampling_ratio/mean": 1.0027551651000977, "sampling/importance_sampling_ratio/max": 1.2980602979660034, "entropy": 0.043996233493089676, "clip_ratio/low_mean": 0.005122950533404946, "clip_ratio/low_min": 0.005122950533404946, "clip_ratio/high_mean": 0.0010245901066809893, "clip_ratio/high_max": 0.0010245901066809893, "clip_ratio/region_mean": 0.006147540640085936, "reward_total_mean": 0.8717328310012817, "reward_meter_mean": 0.9962645173072815, "reward_meter_std": 4.681191057898104e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.8717328310012817, "reward_total_composite_std": 0.050349388271570206} {"timestamp_utc": "2026-04-12T01:15:37Z", "mode": "train", "global_step": 2002, "epoch": 0.08041129453347792, "loss": -0.0194, "grad_norm": 5.45490837097168, "learning_rate": 3.936363636363636e-06, "num_tokens": 4507873.0, "completions/mean_length": 37.375, "completions/min_length": 36.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.375, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.9985835552215576, "rewards/meter/std": 0.00047840786282904446, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9985835552215576, "rewards/total_composite/std": 0.00047840786282904446, "reward": 0.9985835552215576, "reward_std": 0.0004784138291142881, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017835740000009537, "sampling/sampling_logp_difference/max": 0.7125928401947021, "sampling/importance_sampling_ratio/min": 0.49037110805511475, "sampling/importance_sampling_ratio/mean": 1.0080413818359375, "sampling/importance_sampling_ratio/max": 1.2333381175994873, "entropy": 0.1442318456247449, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0032051282469183207, "clip_ratio/high_max": 0.0032051282469183207, "clip_ratio/region_mean": 0.0032051282469183207, "reward_total_mean": 0.9985835552215576, "reward_meter_mean": 0.9985835552215576, "reward_meter_std": 0.00047840786282904446, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9985835552215576, "reward_total_composite_std": 0.00047840786282904446} {"timestamp_utc": "2026-04-12T01:15:42Z", "mode": "train", "global_step": 2003, "epoch": 0.08045146001526288, "loss": -0.0002, "grad_norm": 0.0041707647033035755, "learning_rate": 3.9333333333333335e-06, "num_tokens": 4510281.0, "completions/mean_length": 127.0, "completions/min_length": 127.0, "completions/max_length": 127.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.0, "completions/min_terminated_length": 127.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.9974512457847595, "rewards/meter/std": 1.9482802599668503e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.854958176612854, "rewards/total_composite/std": 1.66965983225964e-05, "reward": 0.854958176612854, "reward_std": 1.669656739977654e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0020607816986739635, "sampling/sampling_logp_difference/max": 1.2046804428100586, "sampling/importance_sampling_ratio/min": 0.2997877895832062, "sampling/importance_sampling_ratio/mean": 1.000118374824524, "sampling/importance_sampling_ratio/max": 1.0359206199645996, "entropy": 0.007686095021199435, "clip_ratio/low_mean": 0.0009842519648373127, "clip_ratio/low_min": 0.0009842519648373127, "clip_ratio/high_mean": 0.0009842519648373127, "clip_ratio/high_max": 0.0009842519648373127, "clip_ratio/region_mean": 0.0019685039296746254, "reward_total_mean": 0.854958176612854, "reward_meter_mean": 0.9974512457847595, "reward_meter_std": 1.9482802599668503e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.854958176612854, "reward_total_composite_std": 1.66965983225964e-05} {"timestamp_utc": "2026-04-12T01:15:47Z", "mode": "train", "global_step": 2004, "epoch": 0.08049162549704783, "loss": 0.0022, "grad_norm": 4.7584662437438965, "learning_rate": 3.930303030303031e-06, "num_tokens": 4511890.0, "completions/mean_length": 38.125, "completions/min_length": 35.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.125, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.999361515045166, "rewards/meter/std": 0.00022483796055894345, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8744252920150757, "rewards/total_composite/std": 0.35332125425338745, "reward": 0.8744252920150757, "reward_std": 0.35332125425338745, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.044290993362665176, "sampling/sampling_logp_difference/max": 1.2952303886413574, "sampling/importance_sampling_ratio/min": 0.273834764957428, "sampling/importance_sampling_ratio/mean": 1.0098013877868652, "sampling/importance_sampling_ratio/max": 1.5466128587722778, "entropy": 0.2946350034326315, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.03295032726600766, "clip_ratio/high_max": 0.03295032726600766, "clip_ratio/region_mean": 0.03295032726600766, "reward_total_mean": 0.8744252920150757, "reward_meter_mean": 0.999361515045166, "reward_meter_std": 0.00022483796055894345, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8744252920150757, "reward_total_composite_std": 0.35332125425338745} {"timestamp_utc": "2026-04-12T01:15:52Z", "mode": "train", "global_step": 2005, "epoch": 0.08053179097883278, "loss": 0.0085, "grad_norm": 2.326425552368164, "learning_rate": 3.927272727272727e-06, "num_tokens": 4514367.0, "completions/mean_length": 121.625, "completions/min_length": 119.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 121.625, "completions/min_terminated_length": 119.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.9959496259689331, "rewards/meter/std": 0.0007494749734178185, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8928571939468384, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.8892085552215576, "rewards/total_composite/std": 0.06532733142375946, "reward": 0.8892085552215576, "reward_std": 0.06532733887434006, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011545258574187756, "sampling/sampling_logp_difference/max": 1.276261329650879, "sampling/importance_sampling_ratio/min": 0.27907872200012207, "sampling/importance_sampling_ratio/mean": 1.0025670528411865, "sampling/importance_sampling_ratio/max": 1.928596019744873, "entropy": 0.07582355523481965, "clip_ratio/low_mean": 0.00512295076623559, "clip_ratio/low_min": 0.00512295076623559, "clip_ratio/high_mean": 0.0072237912099808455, "clip_ratio/high_max": 0.0072237912099808455, "clip_ratio/region_mean": 0.012346741976216435, "reward_total_mean": 0.8892085552215576, "reward_meter_mean": 0.9959496259689331, "reward_meter_std": 0.0007494749734178185, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8928571939468384, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.8892085552215576, "reward_total_composite_std": 0.06532733142375946} {"timestamp_utc": "2026-04-12T01:15:56Z", "mode": "train", "global_step": 2006, "epoch": 0.08057195646061774, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.9242424242424244e-06, "num_tokens": 4515983.0, "completions/mean_length": 35.0, "completions/min_length": 35.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9979252815246582, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979252815246582, "rewards/total_composite/std": 0.0, "reward": 0.9979252815246582, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.004444928839802742, "sampling/sampling_logp_difference/max": 0.16487574577331543, "sampling/importance_sampling_ratio/min": 0.9554311633110046, "sampling/importance_sampling_ratio/mean": 1.0037531852722168, "sampling/importance_sampling_ratio/max": 1.1792465448379517, "entropy": 0.04299367079511285, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9979252815246582, "reward_meter_mean": 0.9979252815246582, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9979252815246582, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:16:01Z", "mode": "train", "global_step": 2007, "epoch": 0.08061212194240269, "loss": 0.0055, "grad_norm": 3.622284412384033, "learning_rate": 3.921212121212122e-06, "num_tokens": 4518013.0, "completions/mean_length": 74.75, "completions/min_length": 71.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.75, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9929636716842651, "rewards/meter/std": 0.008374504745006561, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9929636716842651, "rewards/total_composite/std": 0.008374504745006561, "reward": 0.9929636716842651, "reward_std": 0.008374514989554882, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04201868921518326, "sampling/sampling_logp_difference/max": 1.1939210891723633, "sampling/importance_sampling_ratio/min": 0.30303072929382324, "sampling/importance_sampling_ratio/mean": 1.006500244140625, "sampling/importance_sampling_ratio/max": 1.6419928073883057, "entropy": 0.378201762214303, "clip_ratio/low_mean": 0.00826923071872443, "clip_ratio/low_min": 0.00826923071872443, "clip_ratio/high_mean": 0.026706379372626543, "clip_ratio/high_max": 0.026706379372626543, "clip_ratio/region_mean": 0.03497561009135097, "reward_total_mean": 0.9929636716842651, "reward_meter_mean": 0.9929636716842651, "reward_meter_std": 0.008374504745006561, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9929636716842651, "reward_total_composite_std": 0.008374504745006561} {"timestamp_utc": "2026-04-12T01:16:09Z", "mode": "train", "global_step": 2008, "epoch": 0.08065228742418765, "loss": 0.0072, "grad_norm": 1.4125186204910278, "learning_rate": 3.918181818181819e-06, "num_tokens": 4522347.0, "completions/mean_length": 324.75, "completions/min_length": 309.0, "completions/max_length": 337.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 324.75, "completions/min_terminated_length": 309.0, "completions/max_terminated_length": 337.0, "rewards/meter/mean": 0.7302596569061279, "rewards/meter/std": 0.25050464272499084, "rewards/count_adherence/mean": 0.7272727489471436, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7244791984558105, "rewards/repeat_penalty/std": 0.03113570623099804, "rewards/total_composite/mean": 0.38363176584243774, "rewards/total_composite/std": 0.13244113326072693, "reward": 0.38363176584243774, "reward_std": 0.13244113326072693, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03116171434521675, "sampling/sampling_logp_difference/max": 13.405669212341309, "sampling/importance_sampling_ratio/min": 1.5065788829815574e-06, "sampling/importance_sampling_ratio/mean": 1.0025417804718018, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1241519721224904, "clip_ratio/low_mean": 0.009641719050705433, "clip_ratio/low_min": 0.009641719050705433, "clip_ratio/high_mean": 0.007035914051812142, "clip_ratio/high_max": 0.007035914051812142, "clip_ratio/region_mean": 0.016677633102517575, "reward_total_mean": 0.38363176584243774, "reward_meter_mean": 0.7302596569061279, "reward_meter_std": 0.25050464272499084, "reward_count_adherence_mean": 0.7272727489471436, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7244791984558105, "reward_repeat_penalty_std": 0.03113570623099804, "reward_total_composite_mean": 0.38363176584243774, "reward_total_composite_std": 0.13244113326072693} {"timestamp_utc": "2026-04-12T01:16:14Z", "mode": "train", "global_step": 2009, "epoch": 0.0806924529059726, "loss": -0.0129, "grad_norm": 2.9392364025115967, "learning_rate": 3.915151515151515e-06, "num_tokens": 4524064.0, "completions/mean_length": 57.625, "completions/min_length": 56.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.625, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9944944977760315, "rewards/meter/std": 0.0007702266448177397, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9944944977760315, "rewards/total_composite/std": 0.0007702266448177397, "reward": 0.9944944977760315, "reward_std": 0.0007702277507632971, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011414974927902222, "sampling/sampling_logp_difference/max": 0.5186891555786133, "sampling/importance_sampling_ratio/min": 0.5953003764152527, "sampling/importance_sampling_ratio/mean": 1.0033068656921387, "sampling/importance_sampling_ratio/max": 1.263137936592102, "entropy": 0.08374510239809752, "clip_ratio/low_mean": 0.0022321429569274187, "clip_ratio/low_min": 0.0022321429569274187, "clip_ratio/high_mean": 0.006465517217293382, "clip_ratio/high_max": 0.006465517217293382, "clip_ratio/region_mean": 0.0086976601742208, "reward_total_mean": 0.9944944977760315, "reward_meter_mean": 0.9944944977760315, "reward_meter_std": 0.0007702266448177397, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9944944977760315, "reward_total_composite_std": 0.0007702266448177397} {"timestamp_utc": "2026-04-12T01:16:19Z", "mode": "train", "global_step": 2010, "epoch": 0.08073261838775755, "loss": 0.008, "grad_norm": 2.544851303100586, "learning_rate": 3.912121212121213e-06, "num_tokens": 4526204.0, "completions/mean_length": 103.5, "completions/min_length": 103.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.5, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.8996114730834961, "rewards/meter/std": 0.042233821004629135, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8996114730834961, "rewards/total_composite/std": 0.042233821004629135, "reward": 0.8996114730834961, "reward_std": 0.04223383218050003, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016247009858489037, "sampling/sampling_logp_difference/max": 1.265535831451416, "sampling/importance_sampling_ratio/min": 0.28208810091018677, "sampling/importance_sampling_ratio/mean": 0.9998170137405396, "sampling/importance_sampling_ratio/max": 1.8066370487213135, "entropy": 0.08177287224680185, "clip_ratio/low_mean": 0.005918904324062169, "clip_ratio/low_min": 0.005918904324062169, "clip_ratio/high_mean": 0.00849514571018517, "clip_ratio/high_max": 0.00849514571018517, "clip_ratio/region_mean": 0.014414050034247339, "reward_total_mean": 0.8996114730834961, "reward_meter_mean": 0.8996114730834961, "reward_meter_std": 0.042233821004629135, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8996114730834961, "reward_total_composite_std": 0.042233821004629135} {"timestamp_utc": "2026-04-12T01:16:24Z", "mode": "train", "global_step": 2011, "epoch": 0.08077278386954252, "loss": 0.0002, "grad_norm": 6.366806983947754, "learning_rate": 3.90909090909091e-06, "num_tokens": 4528044.0, "completions/mean_length": 74.0, "completions/min_length": 71.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.0, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9870051145553589, "rewards/meter/std": 0.02932342328131199, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9870051145553589, "rewards/total_composite/std": 0.02932342328131199, "reward": 0.9870051145553589, "reward_std": 0.029323413968086243, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04717674106359482, "sampling/sampling_logp_difference/max": 1.317805290222168, "sampling/importance_sampling_ratio/min": 0.2677222192287445, "sampling/importance_sampling_ratio/mean": 1.0154587030410767, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.35354609973728657, "clip_ratio/low_mean": 0.0033783784601837397, "clip_ratio/low_min": 0.0033783784601837397, "clip_ratio/high_mean": 0.028711115941405296, "clip_ratio/high_max": 0.028711115941405296, "clip_ratio/region_mean": 0.032089494401589036, "reward_total_mean": 0.9870051145553589, "reward_meter_mean": 0.9870051145553589, "reward_meter_std": 0.02932342328131199, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9870051145553589, "reward_total_composite_std": 0.02932342328131199} {"timestamp_utc": "2026-04-12T01:16:28Z", "mode": "train", "global_step": 2012, "epoch": 0.08081294935132748, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.906060606060606e-06, "num_tokens": 4529676.0, "completions/mean_length": 33.0, "completions/min_length": 33.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9990787506103516, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990787506103516, "rewards/total_composite/std": 0.0, "reward": 0.9990787506103516, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00452659884467721, "sampling/sampling_logp_difference/max": 0.21105718612670898, "sampling/importance_sampling_ratio/min": 0.8097277879714966, "sampling/importance_sampling_ratio/mean": 1.0008872747421265, "sampling/importance_sampling_ratio/max": 1.086820363998413, "entropy": 0.025131846428848803, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9990787506103516, "reward_meter_mean": 0.9990787506103516, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990787506103516, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:16:33Z", "mode": "train", "global_step": 2013, "epoch": 0.08085311483311243, "loss": 0.0012, "grad_norm": 1.6898747682571411, "learning_rate": 3.9030303030303035e-06, "num_tokens": 4531806.0, "completions/mean_length": 89.25, "completions/min_length": 88.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.25, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.995819091796875, "rewards/meter/std": 0.0001791357935871929, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.995819091796875, "rewards/total_composite/std": 0.0001791357935871929, "reward": 0.995819091796875, "reward_std": 0.00017913819465320557, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015769556164741516, "sampling/sampling_logp_difference/max": 1.6130731105804443, "sampling/importance_sampling_ratio/min": 0.19927427172660828, "sampling/importance_sampling_ratio/mean": 1.0000059604644775, "sampling/importance_sampling_ratio/max": 1.4373438358306885, "entropy": 0.09196147602051497, "clip_ratio/low_mean": 0.009800249827094376, "clip_ratio/low_min": 0.009800249827094376, "clip_ratio/high_mean": 0.005494505632668734, "clip_ratio/high_max": 0.005494505632668734, "clip_ratio/region_mean": 0.01529475545976311, "reward_total_mean": 0.995819091796875, "reward_meter_mean": 0.995819091796875, "reward_meter_std": 0.0001791357935871929, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.995819091796875, "reward_total_composite_std": 0.0001791357935871929} {"timestamp_utc": "2026-04-12T01:16:37Z", "mode": "train", "global_step": 2014, "epoch": 0.08089328031489738, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.900000000000001e-06, "num_tokens": 4533534.0, "completions/mean_length": 63.0, "completions/min_length": 63.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.0, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9970200061798096, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9970200061798096, "rewards/total_composite/std": 0.0, "reward": 0.9970200061798096, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.002094635274261236, "sampling/sampling_logp_difference/max": 0.07730516046285629, "sampling/importance_sampling_ratio/min": 0.9277377724647522, "sampling/importance_sampling_ratio/mean": 1.0017708539962769, "sampling/importance_sampling_ratio/max": 1.0803717374801636, "entropy": 0.015853506047278643, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9970200061798096, "reward_meter_mean": 0.9970200061798096, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9970200061798096, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:16:42Z", "mode": "train", "global_step": 2015, "epoch": 0.08093344579668234, "loss": 0.0009, "grad_norm": 1.428508996963501, "learning_rate": 3.896969696969697e-06, "num_tokens": 4535555.0, "completions/mean_length": 96.625, "completions/min_length": 96.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.625, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.9975524544715881, "rewards/meter/std": 0.0001665082381805405, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9975524544715881, "rewards/total_composite/std": 0.0001665082381805405, "reward": 0.9975524544715881, "reward_std": 0.00016649799363221973, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018747342750430107, "sampling/sampling_logp_difference/max": 0.7523272037506104, "sampling/importance_sampling_ratio/min": 0.47126853466033936, "sampling/importance_sampling_ratio/mean": 1.0018181800842285, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11758578661829233, "clip_ratio/low_mean": 0.0026041667442768812, "clip_ratio/low_min": 0.0026041667442768812, "clip_ratio/high_mean": 0.015477340901270509, "clip_ratio/high_max": 0.015477340901270509, "clip_ratio/region_mean": 0.01808150764554739, "reward_total_mean": 0.9975524544715881, "reward_meter_mean": 0.9975524544715881, "reward_meter_std": 0.0001665082381805405, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9975524544715881, "reward_total_composite_std": 0.0001665082381805405} {"timestamp_utc": "2026-04-12T01:16:47Z", "mode": "train", "global_step": 2016, "epoch": 0.08097361127846729, "loss": -0.035, "grad_norm": 5.452454566955566, "learning_rate": 3.8939393939393944e-06, "num_tokens": 4536992.0, "completions/mean_length": 32.625, "completions/min_length": 30.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.625, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9971754550933838, "rewards/meter/std": 0.004386701621115208, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9971754550933838, "rewards/total_composite/std": 0.004386701621115208, "reward": 0.9971754550933838, "reward_std": 0.0043867104686796665, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013498242013156414, "sampling/sampling_logp_difference/max": 1.0278760194778442, "sampling/importance_sampling_ratio/min": 0.3577660620212555, "sampling/importance_sampling_ratio/mean": 0.9991469979286194, "sampling/importance_sampling_ratio/max": 1.2878613471984863, "entropy": 0.028148290934041142, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/high_mean": 0.007575757801532745, "clip_ratio/high_max": 0.007575757801532745, "clip_ratio/region_mean": 0.011363636702299118, "reward_total_mean": 0.9971754550933838, "reward_meter_mean": 0.9971754550933838, "reward_meter_std": 0.004386701621115208, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9971754550933838, "reward_total_composite_std": 0.004386701621115208} {"timestamp_utc": "2026-04-12T01:16:52Z", "mode": "train", "global_step": 2017, "epoch": 0.08101377676025225, "loss": 0.0048, "grad_norm": 4.091221332550049, "learning_rate": 3.890909090909092e-06, "num_tokens": 4538962.0, "completions/mean_length": 66.25, "completions/min_length": 65.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.25, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9984210729598999, "rewards/meter/std": 0.0010966768022626638, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9984210729598999, "rewards/total_composite/std": 0.0010966768022626638, "reward": 0.9984210729598999, "reward_std": 0.0010966742411255836, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02255828306078911, "sampling/sampling_logp_difference/max": 1.67164945602417, "sampling/importance_sampling_ratio/min": 0.18793681263923645, "sampling/importance_sampling_ratio/mean": 0.9979748129844666, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06328104855492711, "clip_ratio/low_mean": 0.005681818351149559, "clip_ratio/low_min": 0.005681818351149559, "clip_ratio/high_mean": 0.007604895276017487, "clip_ratio/high_max": 0.007604895276017487, "clip_ratio/region_mean": 0.013286713627167046, "reward_total_mean": 0.9984210729598999, "reward_meter_mean": 0.9984210729598999, "reward_meter_std": 0.0010966768022626638, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9984210729598999, "reward_total_composite_std": 0.0010966768022626638} {"timestamp_utc": "2026-04-12T01:16:57Z", "mode": "train", "global_step": 2018, "epoch": 0.0810539422420372, "loss": 0.0544, "grad_norm": 9.685022354125977, "learning_rate": 3.887878787878788e-06, "num_tokens": 4540736.0, "completions/mean_length": 62.75, "completions/min_length": 57.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.75, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.7036042213439941, "rewards/meter/std": 0.2752836048603058, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.1511857807636261, "rewards/total_composite/mean": 0.5434250831604004, "rewards/total_composite/std": 0.20307792723178864, "reward": 0.5434250831604004, "reward_std": 0.20307794213294983, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0515943206846714, "sampling/sampling_logp_difference/max": 1.9400761127471924, "sampling/importance_sampling_ratio/min": 0.1436930149793625, "sampling/importance_sampling_ratio/mean": 1.0063507556915283, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.28926723077893257, "clip_ratio/low_mean": 0.011172067141160369, "clip_ratio/low_min": 0.011172067141160369, "clip_ratio/high_mean": 0.0331978602334857, "clip_ratio/high_max": 0.0331978602334857, "clip_ratio/region_mean": 0.04436992737464607, "reward_total_mean": 0.5434250831604004, "reward_meter_mean": 0.7036042213439941, "reward_meter_std": 0.2752836048603058, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.1511857807636261, "reward_total_composite_mean": 0.5434250831604004, "reward_total_composite_std": 0.20307792723178864} {"timestamp_utc": "2026-04-12T01:17:02Z", "mode": "train", "global_step": 2019, "epoch": 0.08109410772382215, "loss": -0.0182, "grad_norm": 3.233832359313965, "learning_rate": 3.884848484848485e-06, "num_tokens": 4542988.0, "completions/mean_length": 106.5, "completions/min_length": 100.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.5, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.9943101406097412, "rewards/meter/std": 0.0025435402058064938, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9943101406097412, "rewards/total_composite/std": 0.0025435402058064938, "reward": 0.9943101406097412, "reward_std": 0.0025435429997742176, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03694016858935356, "sampling/sampling_logp_difference/max": 1.4739065170288086, "sampling/importance_sampling_ratio/min": 0.22902902960777283, "sampling/importance_sampling_ratio/mean": 1.010913610458374, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3407053891569376, "clip_ratio/low_mean": 0.014337376691401005, "clip_ratio/low_min": 0.014337376691401005, "clip_ratio/high_mean": 0.013733877101913095, "clip_ratio/high_max": 0.013733877101913095, "clip_ratio/region_mean": 0.0280712537933141, "reward_total_mean": 0.9943101406097412, "reward_meter_mean": 0.9943101406097412, "reward_meter_std": 0.0025435402058064938, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9943101406097412, "reward_total_composite_std": 0.0025435402058064938} {"timestamp_utc": "2026-04-12T01:17:08Z", "mode": "train", "global_step": 2020, "epoch": 0.08113427320560711, "loss": 0.0002, "grad_norm": 2.484889507293701, "learning_rate": 3.881818181818182e-06, "num_tokens": 4545802.0, "completions/mean_length": 173.75, "completions/min_length": 173.0, "completions/max_length": 174.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 173.75, "completions/min_terminated_length": 173.0, "completions/max_terminated_length": 174.0, "rewards/meter/mean": 0.9751757383346558, "rewards/meter/std": 0.06318987905979156, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7124999761581421, "rewards/repeat_penalty/std": 0.0353553481400013, "rewards/total_composite/mean": 0.6928579211235046, "rewards/total_composite/std": 0.015284620225429535, "reward": 0.6928579211235046, "reward_std": 0.015284637920558453, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013370201922953129, "sampling/sampling_logp_difference/max": 1.3443384170532227, "sampling/importance_sampling_ratio/min": 0.2607121467590332, "sampling/importance_sampling_ratio/mean": 1.0012462139129639, "sampling/importance_sampling_ratio/max": 1.7495958805084229, "entropy": 0.07485386310145259, "clip_ratio/low_mean": 0.0007225433364510536, "clip_ratio/low_min": 0.0007225433364510536, "clip_ratio/high_mean": 0.01079247216694057, "clip_ratio/high_max": 0.01079247216694057, "clip_ratio/region_mean": 0.011515015503391623, "reward_total_mean": 0.6928579211235046, "reward_meter_mean": 0.9751757383346558, "reward_meter_std": 0.06318987905979156, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7124999761581421, "reward_repeat_penalty_std": 0.0353553481400013, "reward_total_composite_mean": 0.6928579211235046, "reward_total_composite_std": 0.015284620225429535} {"timestamp_utc": "2026-04-12T01:17:14Z", "mode": "train", "global_step": 2021, "epoch": 0.08117443868739206, "loss": 0.0001, "grad_norm": 1.3713654279708862, "learning_rate": 3.878787878787879e-06, "num_tokens": 4548138.0, "completions/mean_length": 137.0, "completions/min_length": 137.0, "completions/max_length": 137.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 137.0, "completions/min_terminated_length": 137.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.904876172542572, "rewards/meter/std": 0.0019419160671532154, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7756081223487854, "rewards/total_composite/std": 0.0016644843854010105, "reward": 0.7756081223487854, "reward_std": 0.0016644754214212298, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.001481961109675467, "sampling/sampling_logp_difference/max": 0.2336844801902771, "sampling/importance_sampling_ratio/min": 0.9551554918289185, "sampling/importance_sampling_ratio/mean": 1.0013093948364258, "sampling/importance_sampling_ratio/max": 1.263245940208435, "entropy": 0.011890202295035124, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7756081223487854, "reward_meter_mean": 0.904876172542572, "reward_meter_std": 0.0019419160671532154, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7756081223487854, "reward_total_composite_std": 0.0016644843854010105} {"timestamp_utc": "2026-04-12T01:17:18Z", "mode": "train", "global_step": 2022, "epoch": 0.08121460416917702, "loss": -0.0341, "grad_norm": 6.894834518432617, "learning_rate": 3.875757575757576e-06, "num_tokens": 4549759.0, "completions/mean_length": 34.625, "completions/min_length": 32.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.625, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9927211403846741, "rewards/meter/std": 0.009707700461149216, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9927211403846741, "rewards/total_composite/std": 0.009707700461149216, "reward": 0.9927211403846741, "reward_std": 0.009707717224955559, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02147146500647068, "sampling/sampling_logp_difference/max": 1.2381019592285156, "sampling/importance_sampling_ratio/min": 0.2899340093135834, "sampling/importance_sampling_ratio/mean": 0.9955629110336304, "sampling/importance_sampling_ratio/max": 1.2991583347320557, "entropy": 0.07363486429676414, "clip_ratio/low_mean": 0.011029412038624287, "clip_ratio/low_min": 0.011029412038624287, "clip_ratio/high_mean": 0.014087301678955555, "clip_ratio/high_max": 0.014087301678955555, "clip_ratio/region_mean": 0.02511671371757984, "reward_total_mean": 0.9927211403846741, "reward_meter_mean": 0.9927211403846741, "reward_meter_std": 0.009707700461149216, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9927211403846741, "reward_total_composite_std": 0.009707700461149216} {"timestamp_utc": "2026-04-12T01:17:23Z", "mode": "train", "global_step": 2023, "epoch": 0.08125476965096197, "loss": 0.0061, "grad_norm": 2.7678604125976562, "learning_rate": 3.872727272727273e-06, "num_tokens": 4551755.0, "completions/mean_length": 72.5, "completions/min_length": 72.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.5, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9986792802810669, "rewards/meter/std": 0.000637872377410531, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9986792802810669, "rewards/total_composite/std": 0.000637872377410531, "reward": 0.9986792802810669, "reward_std": 0.0006378723192028701, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03179734945297241, "sampling/sampling_logp_difference/max": 0.8867998123168945, "sampling/importance_sampling_ratio/min": 0.4119720160961151, "sampling/importance_sampling_ratio/mean": 1.0089229345321655, "sampling/importance_sampling_ratio/max": 1.7916842699050903, "entropy": 0.24161814339458942, "clip_ratio/low_mean": 0.005184551002457738, "clip_ratio/low_min": 0.005184551002457738, "clip_ratio/high_mean": 0.006896879756823182, "clip_ratio/high_max": 0.006896879756823182, "clip_ratio/region_mean": 0.01208143075928092, "reward_total_mean": 0.9986792802810669, "reward_meter_mean": 0.9986792802810669, "reward_meter_std": 0.000637872377410531, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9986792802810669, "reward_total_composite_std": 0.000637872377410531} {"timestamp_utc": "2026-04-12T01:17:28Z", "mode": "train", "global_step": 2024, "epoch": 0.08129493513274692, "loss": 0.0087, "grad_norm": 1.6175122261047363, "learning_rate": 3.86969696969697e-06, "num_tokens": 4553953.0, "completions/mean_length": 102.75, "completions/min_length": 100.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.75, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.9968816041946411, "rewards/meter/std": 0.0006491380045190454, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9968816041946411, "rewards/total_composite/std": 0.0006491380045190454, "reward": 0.9968816041946411, "reward_std": 0.0006491419626399875, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010959350503981113, "sampling/sampling_logp_difference/max": 0.9986443519592285, "sampling/importance_sampling_ratio/min": 0.36837852001190186, "sampling/importance_sampling_ratio/mean": 0.9999393820762634, "sampling/importance_sampling_ratio/max": 1.40573251247406, "entropy": 0.05600249394774437, "clip_ratio/low_mean": 0.0024038462433964014, "clip_ratio/low_min": 0.0024038462433964014, "clip_ratio/high_mean": 0.006140776677057147, "clip_ratio/high_max": 0.006140776677057147, "clip_ratio/region_mean": 0.008544622920453548, "reward_total_mean": 0.9968816041946411, "reward_meter_mean": 0.9968816041946411, "reward_meter_std": 0.0006491380045190454, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9968816041946411, "reward_total_composite_std": 0.0006491380045190454} {"timestamp_utc": "2026-04-12T01:17:34Z", "mode": "train", "global_step": 2025, "epoch": 0.08133510061453188, "loss": 0.0171, "grad_norm": 3.409259557723999, "learning_rate": 3.866666666666667e-06, "num_tokens": 4556516.0, "completions/mean_length": 150.375, "completions/min_length": 139.0, "completions/max_length": 160.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 150.375, "completions/min_terminated_length": 139.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.890097975730896, "rewards/meter/std": 0.1463252604007721, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.8427009582519531, "rewards/total_composite/std": 0.15601050853729248, "reward": 0.8427009582519531, "reward_std": 0.1560104936361313, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05061601847410202, "sampling/sampling_logp_difference/max": 3.8258116245269775, "sampling/importance_sampling_ratio/min": 0.021800734102725983, "sampling/importance_sampling_ratio/mean": 0.9980630278587341, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2885005250573158, "clip_ratio/low_mean": 0.017962860874831676, "clip_ratio/low_min": 0.017962860874831676, "clip_ratio/high_mean": 0.025219141854904592, "clip_ratio/high_max": 0.025219141854904592, "clip_ratio/region_mean": 0.04318200272973627, "reward_total_mean": 0.8427009582519531, "reward_meter_mean": 0.890097975730896, "reward_meter_std": 0.1463252604007721, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.8427009582519531, "reward_total_composite_std": 0.15601050853729248} {"timestamp_utc": "2026-04-12T01:17:39Z", "mode": "train", "global_step": 2026, "epoch": 0.08137526609631683, "loss": 0.0053, "grad_norm": 2.6563684940338135, "learning_rate": 3.863636363636364e-06, "num_tokens": 4558243.0, "completions/mean_length": 66.875, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.875, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9982448220252991, "rewards/meter/std": 0.0003502867475617677, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9982448220252991, "rewards/total_composite/std": 0.0003502867475617677, "reward": 0.9982448220252991, "reward_std": 0.00035030534490942955, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01347498781979084, "sampling/sampling_logp_difference/max": 0.7937655448913574, "sampling/importance_sampling_ratio/min": 0.4521390497684479, "sampling/importance_sampling_ratio/mean": 0.9976450800895691, "sampling/importance_sampling_ratio/max": 1.3493366241455078, "entropy": 0.038185699842870235, "clip_ratio/low_mean": 0.0037313431967049837, "clip_ratio/low_min": 0.0037313431967049837, "clip_ratio/high_mean": 0.003759611048735678, "clip_ratio/high_max": 0.003759611048735678, "clip_ratio/region_mean": 0.007490954245440662, "reward_total_mean": 0.9982448220252991, "reward_meter_mean": 0.9982448220252991, "reward_meter_std": 0.0003502867475617677, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9982448220252991, "reward_total_composite_std": 0.0003502867475617677} {"timestamp_utc": "2026-04-12T01:17:47Z", "mode": "train", "global_step": 2027, "epoch": 0.08141543157810179, "loss": 0.0071, "grad_norm": 2.305088758468628, "learning_rate": 3.860606060606061e-06, "num_tokens": 4562915.0, "completions/mean_length": 340.0, "completions/min_length": 333.0, "completions/max_length": 342.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 340.0, "completions/min_terminated_length": 333.0, "completions/max_terminated_length": 342.0, "rewards/meter/mean": 0.9870275259017944, "rewards/meter/std": 0.0033544274047017097, "rewards/count_adherence/mean": 0.701923131942749, "rewards/count_adherence/std": 0.027196412906050682, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7379385828971863, "rewards/repeat_penalty/std": 0.02510136552155018, "rewards/total_composite/mean": 0.5113018751144409, "rewards/total_composite/std": 0.02731543779373169, "reward": 0.5113018751144409, "reward_std": 0.027315424755215645, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013775442726910114, "sampling/sampling_logp_difference/max": 1.7148866653442383, "sampling/importance_sampling_ratio/min": 0.17998412251472473, "sampling/importance_sampling_ratio/mean": 1.0029667615890503, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06602455815300345, "clip_ratio/low_mean": 0.005135256564244628, "clip_ratio/low_min": 0.005135256564244628, "clip_ratio/high_mean": 0.0037086496595293283, "clip_ratio/high_max": 0.0037086496595293283, "clip_ratio/region_mean": 0.008843906223773956, "reward_total_mean": 0.5113018751144409, "reward_meter_mean": 0.9870275259017944, "reward_meter_std": 0.0033544274047017097, "reward_count_adherence_mean": 0.701923131942749, "reward_count_adherence_std": 0.027196412906050682, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7379385828971863, "reward_repeat_penalty_std": 0.02510136552155018, "reward_total_composite_mean": 0.5113018751144409, "reward_total_composite_std": 0.02731543779373169} {"timestamp_utc": "2026-04-12T01:17:52Z", "mode": "train", "global_step": 2028, "epoch": 0.08145559705988674, "loss": 0.0109, "grad_norm": 4.434993743896484, "learning_rate": 3.857575757575758e-06, "num_tokens": 4564619.0, "completions/mean_length": 69.0, "completions/min_length": 68.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.0, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9325088262557983, "rewards/meter/std": 0.02886304259300232, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9325088262557983, "rewards/total_composite/std": 0.02886304259300232, "reward": 0.9325088262557983, "reward_std": 0.02886303886771202, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018320947885513306, "sampling/sampling_logp_difference/max": 0.9960315227508545, "sampling/importance_sampling_ratio/min": 0.36934226751327515, "sampling/importance_sampling_ratio/mean": 0.9968265295028687, "sampling/importance_sampling_ratio/max": 1.2660948038101196, "entropy": 0.06549654342234135, "clip_ratio/low_mean": 0.0017857142956927419, "clip_ratio/low_min": 0.0017857142956927419, "clip_ratio/high_mean": 0.007299659075215459, "clip_ratio/high_max": 0.007299659075215459, "clip_ratio/region_mean": 0.0090853733709082, "reward_total_mean": 0.9325088262557983, "reward_meter_mean": 0.9325088262557983, "reward_meter_std": 0.02886304259300232, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9325088262557983, "reward_total_composite_std": 0.02886304259300232} {"timestamp_utc": "2026-04-12T01:17:57Z", "mode": "train", "global_step": 2029, "epoch": 0.08149576254167169, "loss": -0.0021, "grad_norm": 0.44382914900779724, "learning_rate": 3.8545454545454545e-06, "num_tokens": 4566681.0, "completions/mean_length": 102.75, "completions/min_length": 102.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.75, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.9243794083595276, "rewards/meter/std": 0.014914168044924736, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9243794083595276, "rewards/total_composite/std": 0.014914168044924736, "reward": 0.9243794083595276, "reward_std": 0.01491414662450552, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004690106958150864, "sampling/sampling_logp_difference/max": 0.31058406829833984, "sampling/importance_sampling_ratio/min": 0.733018696308136, "sampling/importance_sampling_ratio/mean": 1.002354621887207, "sampling/importance_sampling_ratio/max": 1.119174838066101, "entropy": 0.04349301569163799, "clip_ratio/low_mean": 0.0036526747280731797, "clip_ratio/low_min": 0.0036526747280731797, "clip_ratio/high_mean": 0.0012135922443121672, "clip_ratio/high_max": 0.0012135922443121672, "clip_ratio/region_mean": 0.004866266972385347, "reward_total_mean": 0.9243794083595276, "reward_meter_mean": 0.9243794083595276, "reward_meter_std": 0.014914168044924736, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9243794083595276, "reward_total_composite_std": 0.014914168044924736} {"timestamp_utc": "2026-04-12T01:18:02Z", "mode": "train", "global_step": 2030, "epoch": 0.08153592802345665, "loss": 0.3073, "grad_norm": 10.0611572265625, "learning_rate": 3.851515151515152e-06, "num_tokens": 4568259.0, "completions/mean_length": 47.25, "completions/min_length": 37.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.25, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9984946846961975, "rewards/meter/std": 0.0014516202500090003, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.4629100561141968, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7492817640304565, "rewards/total_composite/std": 0.46246692538261414, "reward": 0.7492817640304565, "reward_std": 0.46246692538261414, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.057400476187467575, "sampling/sampling_logp_difference/max": 2.0996389389038086, "sampling/importance_sampling_ratio/min": 0.12250065058469772, "sampling/importance_sampling_ratio/mean": 1.003790259361267, "sampling/importance_sampling_ratio/max": 1.675675392150879, "entropy": 0.3015878004953265, "clip_ratio/low_mean": 0.013769977260380983, "clip_ratio/low_min": 0.013769977260380983, "clip_ratio/high_mean": 0.019332326017320156, "clip_ratio/high_max": 0.019332326017320156, "clip_ratio/region_mean": 0.03310230327770114, "reward_total_mean": 0.7492817640304565, "reward_meter_mean": 0.9984946846961975, "reward_meter_std": 0.0014516202500090003, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.4629100561141968, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7492817640304565, "reward_total_composite_std": 0.46246692538261414} {"timestamp_utc": "2026-04-12T01:18:07Z", "mode": "train", "global_step": 2031, "epoch": 0.0815760935052416, "loss": -0.004, "grad_norm": 3.1920166015625, "learning_rate": 3.848484848484848e-06, "num_tokens": 4570662.0, "completions/mean_length": 129.375, "completions/min_length": 128.0, "completions/max_length": 134.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 129.375, "completions/min_terminated_length": 128.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.9977312684059143, "rewards/meter/std": 0.00020651114755310118, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8928571939468384, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.8908217549324036, "rewards/total_composite/std": 0.06581787765026093, "reward": 0.8908217549324036, "reward_std": 0.06581789255142212, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01242055743932724, "sampling/sampling_logp_difference/max": 1.437847375869751, "sampling/importance_sampling_ratio/min": 0.23743832111358643, "sampling/importance_sampling_ratio/mean": 1.0028420686721802, "sampling/importance_sampling_ratio/max": 1.6425533294677734, "entropy": 0.06218431191518903, "clip_ratio/low_mean": 0.011571898590773344, "clip_ratio/low_min": 0.011571898590773344, "clip_ratio/high_mean": 0.004751632455736399, "clip_ratio/high_max": 0.004751632455736399, "clip_ratio/region_mean": 0.016323531046509743, "reward_total_mean": 0.8908217549324036, "reward_meter_mean": 0.9977312684059143, "reward_meter_std": 0.00020651114755310118, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8928571939468384, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.8908217549324036, "reward_total_composite_std": 0.06581787765026093} {"timestamp_utc": "2026-04-12T01:18:12Z", "mode": "train", "global_step": 2032, "epoch": 0.08161625898702655, "loss": 0.0009, "grad_norm": 1.8542088270187378, "learning_rate": 3.8454545454545454e-06, "num_tokens": 4572734.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9978283643722534, "rewards/meter/std": 0.00016764015890657902, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978283643722534, "rewards/total_composite/std": 0.00016764015890657902, "reward": 0.9978283643722534, "reward_std": 0.00016763212624937296, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011215658858418465, "sampling/sampling_logp_difference/max": 0.35844850540161133, "sampling/importance_sampling_ratio/min": 0.6987596154212952, "sampling/importance_sampling_ratio/mean": 1.0011532306671143, "sampling/importance_sampling_ratio/max": 1.2213056087493896, "entropy": 0.06596994632855058, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/high_mean": 0.009469697251915932, "clip_ratio/high_max": 0.009469697251915932, "clip_ratio/region_mean": 0.013257576152682304, "reward_total_mean": 0.9978283643722534, "reward_meter_mean": 0.9978283643722534, "reward_meter_std": 0.00016764015890657902, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9978283643722534, "reward_total_composite_std": 0.00016764015890657902} {"timestamp_utc": "2026-04-12T01:18:18Z", "mode": "train", "global_step": 2033, "epoch": 0.08165642446881151, "loss": 0.0323, "grad_norm": 3.462294578552246, "learning_rate": 3.842424242424243e-06, "num_tokens": 4575273.0, "completions/mean_length": 148.375, "completions/min_length": 136.0, "completions/max_length": 163.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 148.375, "completions/min_terminated_length": 136.0, "completions/max_terminated_length": 163.0, "rewards/meter/mean": 0.91922527551651, "rewards/meter/std": 0.09434106945991516, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.910714328289032, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.8339765071868896, "rewards/total_composite/std": 0.08177480101585388, "reward": 0.8339765071868896, "reward_std": 0.08177479356527328, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04417654499411583, "sampling/sampling_logp_difference/max": 1.6957603693008423, "sampling/importance_sampling_ratio/min": 0.18345968425273895, "sampling/importance_sampling_ratio/mean": 1.0021495819091797, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.24012545123696327, "clip_ratio/low_mean": 0.010510046035051346, "clip_ratio/low_min": 0.010510046035051346, "clip_ratio/high_mean": 0.023466870421543717, "clip_ratio/high_max": 0.023466870421543717, "clip_ratio/region_mean": 0.03397691645659506, "reward_total_mean": 0.8339765071868896, "reward_meter_mean": 0.91922527551651, "reward_meter_std": 0.09434106945991516, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.910714328289032, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.8339765071868896, "reward_total_composite_std": 0.08177480101585388} {"timestamp_utc": "2026-04-12T01:18:23Z", "mode": "train", "global_step": 2034, "epoch": 0.08169658995059646, "loss": 0.0002, "grad_norm": 0.040686190128326416, "learning_rate": 3.839393939393939e-06, "num_tokens": 4576945.0, "completions/mean_length": 63.0, "completions/min_length": 63.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.0, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.997020959854126, "rewards/meter/std": 2.6552993404038716e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.997020959854126, "rewards/total_composite/std": 2.6552993404038716e-06, "reward": 0.997020959854126, "reward_std": 2.661313374119345e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0038556025829166174, "sampling/sampling_logp_difference/max": 0.41426944732666016, "sampling/importance_sampling_ratio/min": 0.6608229279518127, "sampling/importance_sampling_ratio/mean": 1.0010058879852295, "sampling/importance_sampling_ratio/max": 1.0738173723220825, "entropy": 0.024745611241087317, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.997020959854126, "reward_meter_mean": 0.997020959854126, "reward_meter_std": 2.6552993404038716e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.997020959854126, "reward_total_composite_std": 2.6552993404038716e-06} {"timestamp_utc": "2026-04-12T01:18:28Z", "mode": "train", "global_step": 2035, "epoch": 0.08173675543238142, "loss": 0.0137, "grad_norm": 5.467068195343018, "learning_rate": 3.836363636363636e-06, "num_tokens": 4578843.0, "completions/mean_length": 74.25, "completions/min_length": 73.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.25, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.8532536625862122, "rewards/meter/std": 0.21336820721626282, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7550032138824463, "rewards/total_composite/std": 0.3712865710258484, "reward": 0.7550032138824463, "reward_std": 0.3712865710258484, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04074222594499588, "sampling/sampling_logp_difference/max": 1.6433086395263672, "sampling/importance_sampling_ratio/min": 0.1933393031358719, "sampling/importance_sampling_ratio/mean": 1.005318522453308, "sampling/importance_sampling_ratio/max": 1.8733973503112793, "entropy": 0.2636965289711952, "clip_ratio/low_mean": 0.008355855825357139, "clip_ratio/low_min": 0.008355855825357139, "clip_ratio/high_mean": 0.0169863011687994, "clip_ratio/high_max": 0.0169863011687994, "clip_ratio/region_mean": 0.02534215699415654, "reward_total_mean": 0.7550032138824463, "reward_meter_mean": 0.8532536625862122, "reward_meter_std": 0.21336820721626282, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7550032138824463, "reward_total_composite_std": 0.3712865710258484} {"timestamp_utc": "2026-04-12T01:18:33Z", "mode": "train", "global_step": 2036, "epoch": 0.08177692091416637, "loss": 0.003, "grad_norm": 1.754205584526062, "learning_rate": 3.833333333333334e-06, "num_tokens": 4581108.0, "completions/mean_length": 108.125, "completions/min_length": 105.0, "completions/max_length": 110.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.125, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 110.0, "rewards/meter/mean": 0.9986991882324219, "rewards/meter/std": 0.0005537345423363149, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9737250804901123, "rewards/total_composite/std": 0.07053209096193314, "reward": 0.9737250804901123, "reward_std": 0.07053209841251373, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02059764787554741, "sampling/sampling_logp_difference/max": 0.9686670303344727, "sampling/importance_sampling_ratio/min": 0.37958869338035583, "sampling/importance_sampling_ratio/mean": 1.00703763961792, "sampling/importance_sampling_ratio/max": 1.7996537685394287, "entropy": 0.21733380481600761, "clip_ratio/low_mean": 0.0022935778833925724, "clip_ratio/low_min": 0.0022935778833925724, "clip_ratio/high_mean": 0.016197497956454754, "clip_ratio/high_max": 0.016197497956454754, "clip_ratio/region_mean": 0.018491075839847326, "reward_total_mean": 0.9737250804901123, "reward_meter_mean": 0.9986991882324219, "reward_meter_std": 0.0005537345423363149, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9737250804901123, "reward_total_composite_std": 0.07053209096193314} {"timestamp_utc": "2026-04-12T01:18:38Z", "mode": "train", "global_step": 2037, "epoch": 0.08181708639595132, "loss": 0.0, "grad_norm": 0.042626433074474335, "learning_rate": 3.830303030303031e-06, "num_tokens": 4582996.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9979811906814575, "rewards/meter/std": 3.9898877730593085e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979811906814575, "rewards/total_composite/std": 3.9898877730593085e-06, "reward": 0.9979811906814575, "reward_std": 4.0090494621836115e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00694712670519948, "sampling/sampling_logp_difference/max": 0.40966981649398804, "sampling/importance_sampling_ratio/min": 0.6638693809509277, "sampling/importance_sampling_ratio/mean": 1.0025206804275513, "sampling/importance_sampling_ratio/max": 1.3170571327209473, "entropy": 0.040048055816441774, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/high_mean": 0.0018939394503831863, "clip_ratio/high_max": 0.0018939394503831863, "clip_ratio/region_mean": 0.005681818351149559, "reward_total_mean": 0.9979811906814575, "reward_meter_mean": 0.9979811906814575, "reward_meter_std": 3.9898877730593085e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9979811906814575, "reward_total_composite_std": 3.9898877730593085e-06} {"timestamp_utc": "2026-04-12T01:18:43Z", "mode": "train", "global_step": 2038, "epoch": 0.08185725187773628, "loss": 0.0002, "grad_norm": 0.05830654874444008, "learning_rate": 3.827272727272728e-06, "num_tokens": 4585172.0, "completions/mean_length": 97.0, "completions/min_length": 97.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.0, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.9979192614555359, "rewards/meter/std": 5.3105509323359e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979192614555359, "rewards/total_composite/std": 5.3105509323359e-06, "reward": 0.9979192614555359, "reward_std": 5.296794824971585e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0037312970962375402, "sampling/sampling_logp_difference/max": 0.49465620517730713, "sampling/importance_sampling_ratio/min": 0.6097805500030518, "sampling/importance_sampling_ratio/mean": 1.0008537769317627, "sampling/importance_sampling_ratio/max": 1.242266297340393, "entropy": 0.025324456859380007, "clip_ratio/low_mean": 0.0012886597542092204, "clip_ratio/low_min": 0.0012886597542092204, "clip_ratio/high_mean": 0.0038659792626276612, "clip_ratio/high_max": 0.0038659792626276612, "clip_ratio/region_mean": 0.005154639016836882, "reward_total_mean": 0.9979192614555359, "reward_meter_mean": 0.9979192614555359, "reward_meter_std": 5.3105509323359e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9979192614555359, "reward_total_composite_std": 5.3105509323359e-06} {"timestamp_utc": "2026-04-12T01:18:48Z", "mode": "train", "global_step": 2039, "epoch": 0.08189741735952123, "loss": 0.0239, "grad_norm": 10.090517044067383, "learning_rate": 3.8242424242424245e-06, "num_tokens": 4587310.0, "completions/mean_length": 110.25, "completions/min_length": 104.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 110.25, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.9942349195480347, "rewards/meter/std": 0.0053013949654996395, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9696316123008728, "rewards/total_composite/std": 0.07374817132949829, "reward": 0.9696316123008728, "reward_std": 0.07374817132949829, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.038861531764268875, "sampling/sampling_logp_difference/max": 1.0281339883804321, "sampling/importance_sampling_ratio/min": 0.3576737940311432, "sampling/importance_sampling_ratio/mean": 1.0104196071624756, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.32149537093937397, "clip_ratio/low_mean": 0.004385964944958687, "clip_ratio/low_min": 0.004385964944958687, "clip_ratio/high_mean": 0.02182656608056277, "clip_ratio/high_max": 0.02182656608056277, "clip_ratio/region_mean": 0.026212531025521457, "reward_total_mean": 0.9696316123008728, "reward_meter_mean": 0.9942349195480347, "reward_meter_std": 0.0053013949654996395, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9696316123008728, "reward_total_composite_std": 0.07374817132949829} {"timestamp_utc": "2026-04-12T01:18:53Z", "mode": "train", "global_step": 2040, "epoch": 0.08193758284130619, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.821212121212122e-06, "num_tokens": 4589110.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9990598559379578, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990598559379578, "rewards/total_composite/std": 0.0, "reward": 0.9990598559379578, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0009253994794562459, "sampling/sampling_logp_difference/max": 0.022413522005081177, "sampling/importance_sampling_ratio/min": 0.9778358340263367, "sampling/importance_sampling_ratio/mean": 1.0006673336029053, "sampling/importance_sampling_ratio/max": 1.0154409408569336, "entropy": 0.009250407747458667, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9990598559379578, "reward_meter_mean": 0.9990598559379578, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990598559379578, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:18:57Z", "mode": "train", "global_step": 2041, "epoch": 0.08197774832309114, "loss": -0.0008, "grad_norm": 4.351480007171631, "learning_rate": 3.818181818181819e-06, "num_tokens": 4590654.0, "completions/mean_length": 39.0, "completions/min_length": 38.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.0, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.9944432973861694, "rewards/meter/std": 0.0023516512010246515, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9944432973861694, "rewards/total_composite/std": 0.0023516512010246515, "reward": 0.9944432973861694, "reward_std": 0.0023516479413956404, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022262930870056152, "sampling/sampling_logp_difference/max": 0.5608975887298584, "sampling/importance_sampling_ratio/min": 0.5875173807144165, "sampling/importance_sampling_ratio/mean": 1.0104680061340332, "sampling/importance_sampling_ratio/max": 1.7522445917129517, "entropy": 0.21922382153570652, "clip_ratio/low_mean": 0.006330128293484449, "clip_ratio/low_min": 0.006330128293484449, "clip_ratio/high_mean": 0.009615384740754962, "clip_ratio/high_max": 0.009615384740754962, "clip_ratio/region_mean": 0.01594551303423941, "reward_total_mean": 0.9944432973861694, "reward_meter_mean": 0.9944432973861694, "reward_meter_std": 0.0023516512010246515, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9944432973861694, "reward_total_composite_std": 0.0023516512010246515} {"timestamp_utc": "2026-04-12T01:19:02Z", "mode": "train", "global_step": 2042, "epoch": 0.0820179138048761, "loss": 0.0002, "grad_norm": 0.49870744347572327, "learning_rate": 3.8151515151515155e-06, "num_tokens": 4592318.0, "completions/mean_length": 58.0, "completions/min_length": 58.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9948880672454834, "rewards/meter/std": 3.073411789955571e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9948880672454834, "rewards/total_composite/std": 3.073411789955571e-05, "reward": 0.9948880672454834, "reward_std": 3.0727183911949396e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004786766599863768, "sampling/sampling_logp_difference/max": 0.8091294765472412, "sampling/importance_sampling_ratio/min": 0.44524553418159485, "sampling/importance_sampling_ratio/mean": 1.001407265663147, "sampling/importance_sampling_ratio/max": 1.819435477256775, "entropy": 0.023268206976354122, "clip_ratio/low_mean": 0.0021551724057644606, "clip_ratio/low_min": 0.0021551724057644606, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0021551724057644606, "reward_total_mean": 0.9948880672454834, "reward_meter_mean": 0.9948880672454834, "reward_meter_std": 3.073411789955571e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9948880672454834, "reward_total_composite_std": 3.073411789955571e-05} {"timestamp_utc": "2026-04-12T01:19:11Z", "mode": "train", "global_step": 2043, "epoch": 0.08205807928666105, "loss": -0.0015, "grad_norm": 2.7885191440582275, "learning_rate": 3.8121212121212127e-06, "num_tokens": 4597255.0, "completions/mean_length": 382.125, "completions/min_length": 379.0, "completions/max_length": 392.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 382.125, "completions/min_terminated_length": 379.0, "completions/max_terminated_length": 392.0, "rewards/meter/mean": 0.9941146373748779, "rewards/meter/std": 0.0013542358065024018, "rewards/count_adherence/mean": 0.6397058963775635, "rewards/count_adherence/std": 0.020797256380319595, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6714285612106323, "rewards/repeat_penalty/std": 0.06172133609652519, "rewards/total_composite/mean": 0.42605897784233093, "rewards/total_composite/std": 0.027482789009809494, "reward": 0.42605897784233093, "reward_std": 0.02748279646039009, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017567651346325874, "sampling/sampling_logp_difference/max": 5.018847465515137, "sampling/importance_sampling_ratio/min": 0.006612143013626337, "sampling/importance_sampling_ratio/mean": 1.0006448030471802, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07445414271205664, "clip_ratio/low_mean": 0.0013192612677812576, "clip_ratio/low_min": 0.0013192612677812576, "clip_ratio/high_mean": 0.01007624133490026, "clip_ratio/high_max": 0.01007624133490026, "clip_ratio/region_mean": 0.011395502602681518, "reward_total_mean": 0.42605897784233093, "reward_meter_mean": 0.9941146373748779, "reward_meter_std": 0.0013542358065024018, "reward_count_adherence_mean": 0.6397058963775635, "reward_count_adherence_std": 0.020797256380319595, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6714285612106323, "reward_repeat_penalty_std": 0.06172133609652519, "reward_total_composite_mean": 0.42605897784233093, "reward_total_composite_std": 0.027482789009809494} {"timestamp_utc": "2026-04-12T01:19:16Z", "mode": "train", "global_step": 2044, "epoch": 0.082098244768446, "loss": -0.018, "grad_norm": 3.922247886657715, "learning_rate": 3.8090909090909095e-06, "num_tokens": 4599133.0, "completions/mean_length": 71.75, "completions/min_length": 69.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.75, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9981293082237244, "rewards/meter/std": 0.0010699051199480891, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981293082237244, "rewards/total_composite/std": 0.0010699051199480891, "reward": 0.9981293082237244, "reward_std": 0.0010698941769078374, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03494240716099739, "sampling/sampling_logp_difference/max": 1.515822410583496, "sampling/importance_sampling_ratio/min": 0.2196274846792221, "sampling/importance_sampling_ratio/mean": 1.008084774017334, "sampling/importance_sampling_ratio/max": 1.6578000783920288, "entropy": 0.23516237549483776, "clip_ratio/low_mean": 0.012303744442760944, "clip_ratio/low_min": 0.012303744442760944, "clip_ratio/high_mean": 0.00866253802087158, "clip_ratio/high_max": 0.00866253802087158, "clip_ratio/region_mean": 0.020966282463632524, "reward_total_mean": 0.9981293082237244, "reward_meter_mean": 0.9981293082237244, "reward_meter_std": 0.0010699051199480891, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981293082237244, "reward_total_composite_std": 0.0010699051199480891} {"timestamp_utc": "2026-04-12T01:19:21Z", "mode": "train", "global_step": 2045, "epoch": 0.08213841025023096, "loss": -0.0019, "grad_norm": 3.759932279586792, "learning_rate": 3.8060606060606064e-06, "num_tokens": 4601598.0, "completions/mean_length": 128.125, "completions/min_length": 127.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.125, "completions/min_terminated_length": 127.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.9989614486694336, "rewards/meter/std": 9.313080954598263e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8928571939468384, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.8919284343719482, "rewards/total_composite/std": 0.06603793799877167, "reward": 0.8919284343719482, "reward_std": 0.06603794544935226, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013363703154027462, "sampling/sampling_logp_difference/max": 1.402510166168213, "sampling/importance_sampling_ratio/min": 0.24597874283790588, "sampling/importance_sampling_ratio/mean": 0.9983217120170593, "sampling/importance_sampling_ratio/max": 1.4929101467132568, "entropy": 0.04036114830523729, "clip_ratio/low_mean": 0.004853463382460177, "clip_ratio/low_min": 0.004853463382460177, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.004853463382460177, "reward_total_mean": 0.8919284343719482, "reward_meter_mean": 0.9989614486694336, "reward_meter_std": 9.313080954598263e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8928571939468384, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.8919284343719482, "reward_total_composite_std": 0.06603793799877167} {"timestamp_utc": "2026-04-12T01:19:26Z", "mode": "train", "global_step": 2046, "epoch": 0.08217857573201591, "loss": -0.0005, "grad_norm": 0.4616295397281647, "learning_rate": 3.803030303030303e-06, "num_tokens": 4603398.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.999051570892334, "rewards/meter/std": 2.334937744308263e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999051570892334, "rewards/total_composite/std": 2.334937744308263e-05, "reward": 0.999051570892334, "reward_std": 2.3349353796220385e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0014573887456208467, "sampling/sampling_logp_difference/max": 0.3329801559448242, "sampling/importance_sampling_ratio/min": 0.7167844176292419, "sampling/importance_sampling_ratio/mean": 1.0000971555709839, "sampling/importance_sampling_ratio/max": 1.0310380458831787, "entropy": 0.007882928592152894, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.999051570892334, "reward_meter_mean": 0.999051570892334, "reward_meter_std": 2.334937744308263e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.999051570892334, "reward_total_composite_std": 2.334937744308263e-05} {"timestamp_utc": "2026-04-12T01:19:31Z", "mode": "train", "global_step": 2047, "epoch": 0.08221874121380086, "loss": -0.0001, "grad_norm": 0.1668306142091751, "learning_rate": 3.8000000000000005e-06, "num_tokens": 4605142.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9979839324951172, "rewards/meter/std": 4.366625034890603e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979839324951172, "rewards/total_composite/std": 4.366625034890603e-06, "reward": 0.9979839324951172, "reward_std": 4.369384441815782e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005823140498250723, "sampling/sampling_logp_difference/max": 0.598487377166748, "sampling/importance_sampling_ratio/min": 0.5496423840522766, "sampling/importance_sampling_ratio/mean": 1.0016472339630127, "sampling/importance_sampling_ratio/max": 1.3960132598876953, "entropy": 0.03522463422268629, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/high_mean": 0.0018939394503831863, "clip_ratio/high_max": 0.0018939394503831863, "clip_ratio/region_mean": 0.0037878789007663727, "reward_total_mean": 0.9979839324951172, "reward_meter_mean": 0.9979839324951172, "reward_meter_std": 4.366625034890603e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9979839324951172, "reward_total_composite_std": 4.366625034890603e-06} {"timestamp_utc": "2026-04-12T01:19:37Z", "mode": "train", "global_step": 2048, "epoch": 0.08225890669558582, "loss": -0.0143, "grad_norm": 1.707572102546692, "learning_rate": 3.7969696969696973e-06, "num_tokens": 4608320.0, "completions/mean_length": 213.25, "completions/min_length": 187.0, "completions/max_length": 221.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 213.25, "completions/min_terminated_length": 187.0, "completions/max_terminated_length": 221.0, "rewards/meter/mean": 0.9987424612045288, "rewards/meter/std": 0.0006651075091212988, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8977272510528564, "rewards/repeat_penalty/std": 0.09009374678134918, "rewards/total_composite/mean": 0.8758239150047302, "rewards/total_composite/std": 0.08225142955780029, "reward": 0.8758239150047302, "reward_std": 0.08225142955780029, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02781507931649685, "sampling/sampling_logp_difference/max": 1.3763513565063477, "sampling/importance_sampling_ratio/min": 0.25249814987182617, "sampling/importance_sampling_ratio/mean": 1.006170392036438, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2369670756161213, "clip_ratio/low_mean": 0.012748151202686131, "clip_ratio/low_min": 0.012748151202686131, "clip_ratio/high_mean": 0.018360439455136657, "clip_ratio/high_max": 0.018360439455136657, "clip_ratio/region_mean": 0.031108590657822788, "reward_total_mean": 0.8758239150047302, "reward_meter_mean": 0.9987424612045288, "reward_meter_std": 0.0006651075091212988, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8977272510528564, "reward_repeat_penalty_std": 0.09009374678134918, "reward_total_composite_mean": 0.8758239150047302, "reward_total_composite_std": 0.08225142955780029} {"timestamp_utc": "2026-04-12T01:19:42Z", "mode": "train", "global_step": 2049, "epoch": 0.08229907217737077, "loss": 0.0002, "grad_norm": 3.169921636581421, "learning_rate": 3.793939393939394e-06, "num_tokens": 4610199.0, "completions/mean_length": 72.875, "completions/min_length": 71.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.875, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9960318207740784, "rewards/meter/std": 0.0027530721854418516, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9960318207740784, "rewards/total_composite/std": 0.0027530721854418516, "reward": 0.9960318207740784, "reward_std": 0.0027530903462320566, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02808314375579357, "sampling/sampling_logp_difference/max": 1.0131583213806152, "sampling/importance_sampling_ratio/min": 0.3630704879760742, "sampling/importance_sampling_ratio/mean": 1.0030927658081055, "sampling/importance_sampling_ratio/max": 1.5734020471572876, "entropy": 0.19552704505622387, "clip_ratio/low_mean": 0.0017123287543654442, "clip_ratio/low_min": 0.0017123287543654442, "clip_ratio/high_mean": 0.00854364933911711, "clip_ratio/high_max": 0.00854364933911711, "clip_ratio/region_mean": 0.010255978093482554, "reward_total_mean": 0.9960318207740784, "reward_meter_mean": 0.9960318207740784, "reward_meter_std": 0.0027530721854418516, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9960318207740784, "reward_total_composite_std": 0.0027530721854418516} {"timestamp_utc": "2026-04-12T01:19:47Z", "mode": "train", "global_step": 2050, "epoch": 0.08233923765915573, "loss": -0.0035, "grad_norm": 2.9366605281829834, "learning_rate": 3.7909090909090914e-06, "num_tokens": 4612079.0, "completions/mean_length": 63.0, "completions/min_length": 62.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.8283913731575012, "rewards/meter/std": 0.1856846958398819, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8283913731575012, "rewards/total_composite/std": 0.1856846958398819, "reward": 0.8283913731575012, "reward_std": 0.1856846958398819, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02338075079023838, "sampling/sampling_logp_difference/max": 2.3884801864624023, "sampling/importance_sampling_ratio/min": 0.09176904708147049, "sampling/importance_sampling_ratio/mean": 1.000585913658142, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15631293877959251, "clip_ratio/low_mean": 0.009891632944345474, "clip_ratio/low_min": 0.009891632944345474, "clip_ratio/high_mean": 0.007843502098694444, "clip_ratio/high_max": 0.007843502098694444, "clip_ratio/region_mean": 0.017735135043039918, "reward_total_mean": 0.8283913731575012, "reward_meter_mean": 0.8283913731575012, "reward_meter_std": 0.1856846958398819, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8283913731575012, "reward_total_composite_std": 0.1856846958398819} {"timestamp_utc": "2026-04-12T01:20:54Z", "mode": "eval", "global_step": 2050, "epoch": 0.08233923765915573, "eval_loss": NaN, "eval_runtime": 67.5758, "eval_samples_per_second": 1.539, "eval_steps_per_second": 0.192, "eval_num_tokens": 4612079.0, "eval_completions/mean_length": 201.2403846153846, "eval_completions/min_length": 63.30769230769231, "eval_completions/max_length": 354.6923076923077, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 201.2403846153846, "eval_completions/min_terminated_length": 63.30769230769231, "eval_completions/max_terminated_length": 354.6923076923077, "eval_rewards/meter/mean": 0.7046224291508014, "eval_rewards/meter/std": 0.42778917917838466, "eval_rewards/count_adherence/mean": 0.9102904154704168, "eval_rewards/count_adherence/std": 0.12148115927210221, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/repeat_penalty/mean": 0.80578757249392, "eval_rewards/repeat_penalty/std": 0.18289457318874505, "eval_rewards/total_composite/mean": 0.5263216747687414, "eval_rewards/total_composite/std": 0.37407297583726734, "eval_reward": 0.5263216747687414, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.013937483756588055, "eval_sampling/sampling_logp_difference/max": 1.54466306246244, "eval_sampling/importance_sampling_ratio/min": 0.2842731796778165, "eval_sampling/importance_sampling_ratio/mean": 1.0030618355824397, "eval_sampling/importance_sampling_ratio/max": 1.4798904473964984, "eval_entropy": 0.12492357309047993, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5263216747687414, "eval_reward_meter_mean": 0.7046224291508014, "eval_reward_meter_std": 0.42778917917838466, "eval_reward_count_adherence_mean": 0.9102904154704168, "eval_reward_count_adherence_std": 0.12148115927210221, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_repeat_penalty_mean": 0.80578757249392, "eval_reward_repeat_penalty_std": 0.18289457318874505, "eval_reward_total_composite_mean": 0.5263216747687414, "eval_reward_total_composite_std": 0.37407297583726734} {"timestamp_utc": "2026-04-12T01:21:05Z", "mode": "train", "global_step": 2051, "epoch": 0.08237940314094068, "loss": -0.0002, "grad_norm": 0.02642267569899559, "learning_rate": 3.7878787878787882e-06, "num_tokens": 4616222.0, "completions/mean_length": 312.875, "completions/min_length": 312.0, "completions/max_length": 313.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 312.875, "completions/min_terminated_length": 312.0, "completions/max_terminated_length": 313.0, "rewards/meter/mean": 0.9973797798156738, "rewards/meter/std": 8.498986062477343e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5789473652839661, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5774303674697876, "rewards/total_composite/std": 4.927758254780201e-06, "reward": 0.5774303674697876, "reward_std": 4.94351661473047e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.001822733087465167, "sampling/sampling_logp_difference/max": 0.5258277654647827, "sampling/importance_sampling_ratio/min": 0.5910658836364746, "sampling/importance_sampling_ratio/mean": 1.0007059574127197, "sampling/importance_sampling_ratio/max": 1.331079125404358, "entropy": 0.01078800146933645, "clip_ratio/low_mean": 0.0003993610152974725, "clip_ratio/low_min": 0.0003993610152974725, "clip_ratio/high_mean": 0.0008000020461622626, "clip_ratio/high_max": 0.0008000020461622626, "clip_ratio/region_mean": 0.001199363061459735, "reward_total_mean": 0.5774303674697876, "reward_meter_mean": 0.9973797798156738, "reward_meter_std": 8.498986062477343e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5789473652839661, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5774303674697876, "reward_total_composite_std": 4.927758254780201e-06} {"timestamp_utc": "2026-04-12T01:21:12Z", "mode": "train", "global_step": 2052, "epoch": 0.08241956862272563, "loss": -0.0126, "grad_norm": 1.5723775625228882, "learning_rate": 3.784848484848485e-06, "num_tokens": 4619661.0, "completions/mean_length": 214.875, "completions/min_length": 208.0, "completions/max_length": 221.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 214.875, "completions/min_terminated_length": 208.0, "completions/max_terminated_length": 221.0, "rewards/meter/mean": 0.9847596883773804, "rewards/meter/std": 0.03142433241009712, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.625, "rewards/repeat_penalty/std": 0.2591308057308197, "rewards/total_composite/mean": 0.6123065948486328, "rewards/total_composite/std": 0.24669423699378967, "reward": 0.6123065948486328, "reward_std": 0.24669425189495087, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023984558880329132, "sampling/sampling_logp_difference/max": 1.9047002792358398, "sampling/importance_sampling_ratio/min": 0.14886726438999176, "sampling/importance_sampling_ratio/mean": 1.0024995803833008, "sampling/importance_sampling_ratio/max": 1.9927300214767456, "entropy": 0.18392128869891167, "clip_ratio/low_mean": 0.0017639786819927394, "clip_ratio/low_min": 0.0017639786819927394, "clip_ratio/high_mean": 0.014248464838601649, "clip_ratio/high_max": 0.014248464838601649, "clip_ratio/region_mean": 0.016012443520594388, "reward_total_mean": 0.6123065948486328, "reward_meter_mean": 0.9847596883773804, "reward_meter_std": 0.03142433241009712, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.625, "reward_repeat_penalty_std": 0.2591308057308197, "reward_total_composite_mean": 0.6123065948486328, "reward_total_composite_std": 0.24669423699378967} {"timestamp_utc": "2026-04-12T01:21:16Z", "mode": "train", "global_step": 2053, "epoch": 0.08245973410451059, "loss": 0.0254, "grad_norm": 7.329649448394775, "learning_rate": 3.781818181818182e-06, "num_tokens": 4621203.0, "completions/mean_length": 38.75, "completions/min_length": 36.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.75, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.9991681575775146, "rewards/meter/std": 0.0005326797836460173, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991681575775146, "rewards/total_composite/std": 0.0005326797836460173, "reward": 0.9991681575775146, "reward_std": 0.0005326886312104762, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02681351639330387, "sampling/sampling_logp_difference/max": 0.8889703750610352, "sampling/importance_sampling_ratio/min": 0.4110788106918335, "sampling/importance_sampling_ratio/mean": 1.006339430809021, "sampling/importance_sampling_ratio/max": 1.5577260255813599, "entropy": 0.16460048034787178, "clip_ratio/low_mean": 0.006253908621147275, "clip_ratio/low_min": 0.006253908621147275, "clip_ratio/high_mean": 0.0065972222946584225, "clip_ratio/high_max": 0.0065972222946584225, "clip_ratio/region_mean": 0.012851130915805697, "reward_total_mean": 0.9991681575775146, "reward_meter_mean": 0.9991681575775146, "reward_meter_std": 0.0005326797836460173, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9991681575775146, "reward_total_composite_std": 0.0005326797836460173} {"timestamp_utc": "2026-04-12T01:21:21Z", "mode": "train", "global_step": 2054, "epoch": 0.08249989958629554, "loss": 0.0004, "grad_norm": 0.35728779435157776, "learning_rate": 3.778787878787879e-06, "num_tokens": 4623051.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.997982382774353, "rewards/meter/std": 1.786020766303409e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.997982382774353, "rewards/total_composite/std": 1.786020766303409e-05, "reward": 0.997982382774353, "reward_std": 1.7854295947472565e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009083742275834084, "sampling/sampling_logp_difference/max": 0.5304985046386719, "sampling/importance_sampling_ratio/min": 0.5883116722106934, "sampling/importance_sampling_ratio/mean": 0.999168872833252, "sampling/importance_sampling_ratio/max": 1.3692970275878906, "entropy": 0.03993460722267628, "clip_ratio/low_mean": 0.007575757801532745, "clip_ratio/low_min": 0.007575757801532745, "clip_ratio/high_mean": 0.007575757801532745, "clip_ratio/high_max": 0.007575757801532745, "clip_ratio/region_mean": 0.01515151560306549, "reward_total_mean": 0.997982382774353, "reward_meter_mean": 0.997982382774353, "reward_meter_std": 1.786020766303409e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.997982382774353, "reward_total_composite_std": 1.786020766303409e-05} {"timestamp_utc": "2026-04-12T01:21:29Z", "mode": "train", "global_step": 2055, "epoch": 0.0825400650680805, "loss": -0.0003, "grad_norm": 2.26758074760437, "learning_rate": 3.775757575757576e-06, "num_tokens": 4627144.0, "completions/mean_length": 315.625, "completions/min_length": 305.0, "completions/max_length": 330.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 315.625, "completions/min_terminated_length": 305.0, "completions/max_terminated_length": 330.0, "rewards/meter/mean": 0.9283133149147034, "rewards/meter/std": 0.12819351255893707, "rewards/count_adherence/mean": 0.862500011920929, "rewards/count_adherence/std": 0.0517548993229866, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8436580896377563, "rewards/repeat_penalty/std": 0.11612330377101898, "rewards/total_composite/mean": 0.6760411262512207, "rewards/total_composite/std": 0.1399042010307312, "reward": 0.6760411262512207, "reward_std": 0.1399042010307312, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0354769341647625, "sampling/sampling_logp_difference/max": 2.784627914428711, "sampling/importance_sampling_ratio/min": 0.061752062290906906, "sampling/importance_sampling_ratio/mean": 1.002432942390442, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.22647083643823862, "clip_ratio/low_mean": 0.006354167824611068, "clip_ratio/low_min": 0.006354167824611068, "clip_ratio/high_mean": 0.026429779594764113, "clip_ratio/high_max": 0.026429779594764113, "clip_ratio/region_mean": 0.03278394741937518, "reward_total_mean": 0.6760411262512207, "reward_meter_mean": 0.9283133149147034, "reward_meter_std": 0.12819351255893707, "reward_count_adherence_mean": 0.862500011920929, "reward_count_adherence_std": 0.0517548993229866, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8436580896377563, "reward_repeat_penalty_std": 0.11612330377101898, "reward_total_composite_mean": 0.6760411262512207, "reward_total_composite_std": 0.1399042010307312} {"timestamp_utc": "2026-04-12T01:21:34Z", "mode": "train", "global_step": 2056, "epoch": 0.08258023054986545, "loss": -0.0133, "grad_norm": 8.520014762878418, "learning_rate": 3.772727272727273e-06, "num_tokens": 4628924.0, "completions/mean_length": 61.5, "completions/min_length": 42.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.5, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.6756709814071655, "rewards/meter/std": 0.25546279549598694, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6134400963783264, "rewards/total_composite/std": 0.22521547973155975, "reward": 0.6134400963783264, "reward_std": 0.22521549463272095, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031044768169522285, "sampling/sampling_logp_difference/max": 2.2168161869049072, "sampling/importance_sampling_ratio/min": 0.10895545780658722, "sampling/importance_sampling_ratio/mean": 1.0012702941894531, "sampling/importance_sampling_ratio/max": 1.7200733423233032, "entropy": 0.2539948094636202, "clip_ratio/low_mean": 0.007753314450383186, "clip_ratio/low_min": 0.007753314450383186, "clip_ratio/high_mean": 0.013611778849735856, "clip_ratio/high_max": 0.013611778849735856, "clip_ratio/region_mean": 0.021365093300119042, "reward_total_mean": 0.6134400963783264, "reward_meter_mean": 0.6756709814071655, "reward_meter_std": 0.25546279549598694, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6134400963783264, "reward_total_composite_std": 0.22521547973155975} {"timestamp_utc": "2026-04-12T01:21:39Z", "mode": "train", "global_step": 2057, "epoch": 0.0826203960316504, "loss": -0.001, "grad_norm": 0.4782918691635132, "learning_rate": 3.76969696969697e-06, "num_tokens": 4630892.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9990358352661133, "rewards/meter/std": 6.781429692637175e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990358352661133, "rewards/total_composite/std": 6.781429692637175e-05, "reward": 0.9990358352661133, "reward_std": 6.782029959140345e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0013893973082304, "sampling/sampling_logp_difference/max": 0.41433000564575195, "sampling/importance_sampling_ratio/min": 0.6607828736305237, "sampling/importance_sampling_ratio/mean": 0.9997783303260803, "sampling/importance_sampling_ratio/max": 1.0163952112197876, "entropy": 0.00615574762923643, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9990358352661133, "reward_meter_mean": 0.9990358352661133, "reward_meter_std": 6.781429692637175e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990358352661133, "reward_total_composite_std": 6.781429692637175e-05} {"timestamp_utc": "2026-04-12T01:21:43Z", "mode": "train", "global_step": 2058, "epoch": 0.08266056151343536, "loss": 0.0, "grad_norm": 0.47224369645118713, "learning_rate": 3.766666666666667e-06, "num_tokens": 4632844.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9979783296585083, "rewards/meter/std": 4.64778822788503e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979783296585083, "rewards/total_composite/std": 4.64778822788503e-05, "reward": 0.9979783296585083, "reward_std": 4.6468860091408715e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004593671765178442, "sampling/sampling_logp_difference/max": 0.4634366035461426, "sampling/importance_sampling_ratio/min": 0.6291179060935974, "sampling/importance_sampling_ratio/mean": 1.0018833875656128, "sampling/importance_sampling_ratio/max": 1.5034801959991455, "entropy": 0.029094983357936144, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9979783296585083, "reward_meter_mean": 0.9979783296585083, "reward_meter_std": 4.64778822788503e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9979783296585083, "reward_total_composite_std": 4.64778822788503e-05} {"timestamp_utc": "2026-04-12T01:21:48Z", "mode": "train", "global_step": 2059, "epoch": 0.08270072699522031, "loss": -0.0039, "grad_norm": 3.3012750148773193, "learning_rate": 3.7636363636363637e-06, "num_tokens": 4634771.0, "completions/mean_length": 65.875, "completions/min_length": 65.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.875, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9989724159240723, "rewards/meter/std": 0.00024742307141423225, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989724159240723, "rewards/total_composite/std": 0.00024742307141423225, "reward": 0.9989724159240723, "reward_std": 0.0002474260691087693, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0033799612428992987, "sampling/sampling_logp_difference/max": 1.322075366973877, "sampling/importance_sampling_ratio/min": 0.2665814757347107, "sampling/importance_sampling_ratio/mean": 0.9990096688270569, "sampling/importance_sampling_ratio/max": 1.0218313932418823, "entropy": 0.0075465834233909845, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9989724159240723, "reward_meter_mean": 0.9989724159240723, "reward_meter_std": 0.00024742307141423225, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9989724159240723, "reward_total_composite_std": 0.00024742307141423225} {"timestamp_utc": "2026-04-12T01:21:53Z", "mode": "train", "global_step": 2060, "epoch": 0.08274089247700527, "loss": -0.0013, "grad_norm": 0.7633002400398254, "learning_rate": 3.7606060606060605e-06, "num_tokens": 4636937.0, "completions/mean_length": 98.75, "completions/min_length": 98.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.75, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.99910569190979, "rewards/meter/std": 4.124943006900139e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.99910569190979, "rewards/total_composite/std": 4.124943006900139e-05, "reward": 0.99910569190979, "reward_std": 4.125862324144691e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0010279221460223198, "sampling/sampling_logp_difference/max": 0.05605363845825195, "sampling/importance_sampling_ratio/min": 0.9609670042991638, "sampling/importance_sampling_ratio/mean": 1.0008070468902588, "sampling/importance_sampling_ratio/max": 1.0576543807983398, "entropy": 0.011058209463953972, "clip_ratio/low_mean": 0.0025510203558951616, "clip_ratio/low_min": 0.0025510203558951616, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0025510203558951616, "reward_total_mean": 0.99910569190979, "reward_meter_mean": 0.99910569190979, "reward_meter_std": 4.124943006900139e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.99910569190979, "reward_total_composite_std": 4.124943006900139e-05} {"timestamp_utc": "2026-04-12T01:21:58Z", "mode": "train", "global_step": 2061, "epoch": 0.08278105795879022, "loss": 0.0046, "grad_norm": 2.503458023071289, "learning_rate": 3.757575757575758e-06, "num_tokens": 4638510.0, "completions/mean_length": 38.625, "completions/min_length": 36.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.625, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.9993641972541809, "rewards/meter/std": 0.0002709543623495847, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993641972541809, "rewards/total_composite/std": 0.0002709543623495847, "reward": 0.9993641972541809, "reward_std": 0.0002709643158596009, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026116767898201942, "sampling/sampling_logp_difference/max": 1.114222526550293, "sampling/importance_sampling_ratio/min": 0.3281702995300293, "sampling/importance_sampling_ratio/mean": 1.0021777153015137, "sampling/importance_sampling_ratio/max": 1.691472053527832, "entropy": 0.1621959926560521, "clip_ratio/low_mean": 0.012820512987673283, "clip_ratio/low_min": 0.012820512987673283, "clip_ratio/high_mean": 0.016136696096509695, "clip_ratio/high_max": 0.016136696096509695, "clip_ratio/region_mean": 0.028957209084182978, "reward_total_mean": 0.9993641972541809, "reward_meter_mean": 0.9993641972541809, "reward_meter_std": 0.0002709543623495847, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993641972541809, "reward_total_composite_std": 0.0002709543623495847} {"timestamp_utc": "2026-04-12T01:22:04Z", "mode": "train", "global_step": 2062, "epoch": 0.08282122344057517, "loss": 0.0007, "grad_norm": 0.12022405117750168, "learning_rate": 3.7545454545454546e-06, "num_tokens": 4640965.0, "completions/mean_length": 127.875, "completions/min_length": 127.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.875, "completions/min_terminated_length": 127.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.9978995323181152, "rewards/meter/std": 1.4925647519703489e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8553425073623657, "rewards/total_composite/std": 1.2786828847310971e-05, "reward": 0.8553425073623657, "reward_std": 1.2798592251783703e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004213304258882999, "sampling/sampling_logp_difference/max": 0.8202402591705322, "sampling/importance_sampling_ratio/min": 0.44032585620880127, "sampling/importance_sampling_ratio/mean": 0.9991137981414795, "sampling/importance_sampling_ratio/max": 1.108210802078247, "entropy": 0.019355892902240157, "clip_ratio/low_mean": 0.0009765625, "clip_ratio/low_min": 0.0009765625, "clip_ratio/high_mean": 0.0029450664296746254, "clip_ratio/high_max": 0.0029450664296746254, "clip_ratio/region_mean": 0.003921628929674625, "reward_total_mean": 0.8553425073623657, "reward_meter_mean": 0.9978995323181152, "reward_meter_std": 1.4925647519703489e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8553425073623657, "reward_total_composite_std": 1.2786828847310971e-05} {"timestamp_utc": "2026-04-12T01:22:09Z", "mode": "train", "global_step": 2063, "epoch": 0.08286138892236013, "loss": -0.0003, "grad_norm": 0.1908394694328308, "learning_rate": 3.7515151515151515e-06, "num_tokens": 4643037.0, "completions/mean_length": 97.0, "completions/min_length": 97.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.0, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.9979283809661865, "rewards/meter/std": 1.0387539987277705e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979283809661865, "rewards/total_composite/std": 1.0387539987277705e-05, "reward": 0.9979283809661865, "reward_std": 1.038749087456381e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0028961896896362305, "sampling/sampling_logp_difference/max": 0.6855463981628418, "sampling/importance_sampling_ratio/min": 0.5038148760795593, "sampling/importance_sampling_ratio/mean": 1.001339077949524, "sampling/importance_sampling_ratio/max": 1.0733686685562134, "entropy": 0.020256070652976632, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9979283809661865, "reward_meter_mean": 0.9979283809661865, "reward_meter_std": 1.0387539987277705e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9979283809661865, "reward_total_composite_std": 1.0387539987277705e-05} {"timestamp_utc": "2026-04-12T01:22:14Z", "mode": "train", "global_step": 2064, "epoch": 0.08290155440414508, "loss": -0.0028, "grad_norm": 0.6916006207466125, "learning_rate": 3.748484848484849e-06, "num_tokens": 4645538.0, "completions/mean_length": 119.625, "completions/min_length": 118.0, "completions/max_length": 120.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 119.625, "completions/min_terminated_length": 118.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.996715784072876, "rewards/meter/std": 0.00013060594210401177, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8543277978897095, "rewards/total_composite/std": 0.00011193905083928257, "reward": 0.8543277978897095, "reward_std": 0.00011193443788215518, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004635249730199575, "sampling/sampling_logp_difference/max": 0.4385652542114258, "sampling/importance_sampling_ratio/min": 0.6449611186981201, "sampling/importance_sampling_ratio/mean": 1.0017560720443726, "sampling/importance_sampling_ratio/max": 1.5008399486541748, "entropy": 0.02747240220196545, "clip_ratio/low_mean": 0.004228386213071644, "clip_ratio/low_min": 0.004228386213071644, "clip_ratio/high_mean": 0.0010416667209938169, "clip_ratio/high_max": 0.0010416667209938169, "clip_ratio/region_mean": 0.005270052934065461, "reward_total_mean": 0.8543277978897095, "reward_meter_mean": 0.996715784072876, "reward_meter_std": 0.00013060594210401177, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8543277978897095, "reward_total_composite_std": 0.00011193905083928257} {"timestamp_utc": "2026-04-12T01:22:19Z", "mode": "train", "global_step": 2065, "epoch": 0.08294171988593003, "loss": -0.0076, "grad_norm": 3.2012574672698975, "learning_rate": 3.745454545454546e-06, "num_tokens": 4647558.0, "completions/mean_length": 74.5, "completions/min_length": 72.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.5, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9918162822723389, "rewards/meter/std": 0.009919598698616028, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9918162822723389, "rewards/total_composite/std": 0.009919598698616028, "reward": 0.9918162822723389, "reward_std": 0.009919599629938602, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023357173427939415, "sampling/sampling_logp_difference/max": 1.084247350692749, "sampling/importance_sampling_ratio/min": 0.33815622329711914, "sampling/importance_sampling_ratio/mean": 1.004847526550293, "sampling/importance_sampling_ratio/max": 1.9737117290496826, "entropy": 0.1982443816959858, "clip_ratio/low_mean": 0.0034722222480922937, "clip_ratio/low_min": 0.0034722222480922937, "clip_ratio/high_mean": 0.016551562468521297, "clip_ratio/high_max": 0.016551562468521297, "clip_ratio/region_mean": 0.02002378471661359, "reward_total_mean": 0.9918162822723389, "reward_meter_mean": 0.9918162822723389, "reward_meter_std": 0.009919598698616028, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9918162822723389, "reward_total_composite_std": 0.009919598698616028} {"timestamp_utc": "2026-04-12T01:22:23Z", "mode": "train", "global_step": 2066, "epoch": 0.08298188536771499, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.742424242424243e-06, "num_tokens": 4649430.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9990598559379578, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990598559379578, "rewards/total_composite/std": 0.0, "reward": 0.9990598559379578, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002762091171462089, "sampling/sampling_logp_difference/max": 0.004217715933918953, "sampling/importance_sampling_ratio/min": 0.997229278087616, "sampling/importance_sampling_ratio/mean": 1.0002522468566895, "sampling/importance_sampling_ratio/max": 1.0042266845703125, "entropy": 0.002455963665852323, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9990598559379578, "reward_meter_mean": 0.9990598559379578, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990598559379578, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:22:32Z", "mode": "train", "global_step": 2067, "epoch": 0.08302205084949994, "loss": -0.0169, "grad_norm": 1.5572625398635864, "learning_rate": 3.73939393939394e-06, "num_tokens": 4654027.0, "completions/mean_length": 381.625, "completions/min_length": 362.0, "completions/max_length": 398.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 381.625, "completions/min_terminated_length": 362.0, "completions/max_terminated_length": 398.0, "rewards/meter/mean": 0.9989120960235596, "rewards/meter/std": 0.0005744565860368311, "rewards/count_adherence/mean": 0.65625, "rewards/count_adherence/std": 0.033407654613256454, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7585839033126831, "rewards/repeat_penalty/std": 0.06556817144155502, "rewards/total_composite/mean": 0.4973776340484619, "rewards/total_composite/std": 0.04983275756239891, "reward": 0.4973776340484619, "reward_std": 0.04983276501297951, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.034968629479408264, "sampling/sampling_logp_difference/max": 14.583293914794922, "sampling/importance_sampling_ratio/min": 4.6404053932747047e-07, "sampling/importance_sampling_ratio/mean": 1.0025993585586548, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2074881698936224, "clip_ratio/low_mean": 0.009894335642457008, "clip_ratio/low_min": 0.009894335642457008, "clip_ratio/high_mean": 0.012662693159654737, "clip_ratio/high_max": 0.012662693159654737, "clip_ratio/region_mean": 0.022557028802111745, "reward_total_mean": 0.4973776340484619, "reward_meter_mean": 0.9989120960235596, "reward_meter_std": 0.0005744565860368311, "reward_count_adherence_mean": 0.65625, "reward_count_adherence_std": 0.033407654613256454, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7585839033126831, "reward_repeat_penalty_std": 0.06556817144155502, "reward_total_composite_mean": 0.4973776340484619, "reward_total_composite_std": 0.04983275756239891} {"timestamp_utc": "2026-04-12T01:22:37Z", "mode": "train", "global_step": 2068, "epoch": 0.0830622163312849, "loss": -0.0084, "grad_norm": 5.365983009338379, "learning_rate": 3.736363636363637e-06, "num_tokens": 4655794.0, "completions/mean_length": 74.875, "completions/min_length": 70.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.875, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9989867210388184, "rewards/meter/std": 0.0009042008896358311, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989867210388184, "rewards/total_composite/std": 0.0009042008896358311, "reward": 0.9989867210388184, "reward_std": 0.000904214452020824, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027786457911133766, "sampling/sampling_logp_difference/max": 0.7996888160705566, "sampling/importance_sampling_ratio/min": 0.44946879148483276, "sampling/importance_sampling_ratio/mean": 1.0074291229248047, "sampling/importance_sampling_ratio/max": 1.8772411346435547, "entropy": 0.22592910565435886, "clip_ratio/low_mean": 0.0035714285913854837, "clip_ratio/low_min": 0.0035714285913854837, "clip_ratio/high_mean": 0.01815969729796052, "clip_ratio/high_max": 0.01815969729796052, "clip_ratio/region_mean": 0.021731125889346004, "reward_total_mean": 0.9989867210388184, "reward_meter_mean": 0.9989867210388184, "reward_meter_std": 0.0009042008896358311, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9989867210388184, "reward_total_composite_std": 0.0009042008896358311} {"timestamp_utc": "2026-04-12T01:22:41Z", "mode": "train", "global_step": 2069, "epoch": 0.08310238181306985, "loss": 0.008, "grad_norm": 3.824120283126831, "learning_rate": 3.7333333333333337e-06, "num_tokens": 4657320.0, "completions/mean_length": 29.75, "completions/min_length": 29.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.75, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.9928488731384277, "rewards/meter/std": 7.0027461333666e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9928488731384277, "rewards/total_composite/std": 7.0027461333666e-05, "reward": 0.9928488731384277, "reward_std": 7.002745405770838e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0071687763556838036, "sampling/sampling_logp_difference/max": 0.6952643394470215, "sampling/importance_sampling_ratio/min": 0.9426794052124023, "sampling/importance_sampling_ratio/mean": 1.008081316947937, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03380009834654629, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9928488731384277, "reward_meter_mean": 0.9928488731384277, "reward_meter_std": 7.0027461333666e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9928488731384277, "reward_total_composite_std": 7.0027461333666e-05} {"timestamp_utc": "2026-04-12T01:22:45Z", "mode": "train", "global_step": 2070, "epoch": 0.0831425472948548, "loss": 0.0041, "grad_norm": 2.178863286972046, "learning_rate": 3.7303030303030306e-06, "num_tokens": 4659028.0, "completions/mean_length": 57.5, "completions/min_length": 56.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.5, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9949300289154053, "rewards/meter/std": 4.729199281428009e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9949300289154053, "rewards/total_composite/std": 4.729199281428009e-05, "reward": 0.9949300289154053, "reward_std": 4.729198190034367e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006639817729592323, "sampling/sampling_logp_difference/max": 0.6818382740020752, "sampling/importance_sampling_ratio/min": 0.5056865215301514, "sampling/importance_sampling_ratio/mean": 1.0030012130737305, "sampling/importance_sampling_ratio/max": 1.951530933380127, "entropy": 0.03393339877948165, "clip_ratio/low_mean": 0.008620689623057842, "clip_ratio/low_min": 0.008620689623057842, "clip_ratio/high_mean": 0.0022321429569274187, "clip_ratio/high_max": 0.0022321429569274187, "clip_ratio/region_mean": 0.010852832579985261, "reward_total_mean": 0.9949300289154053, "reward_meter_mean": 0.9949300289154053, "reward_meter_std": 4.729199281428009e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9949300289154053, "reward_total_composite_std": 4.729199281428009e-05} {"timestamp_utc": "2026-04-12T01:22:50Z", "mode": "train", "global_step": 2071, "epoch": 0.08318271277663976, "loss": 0.0012, "grad_norm": 2.837266206741333, "learning_rate": 3.727272727272728e-06, "num_tokens": 4661038.0, "completions/mean_length": 87.25, "completions/min_length": 86.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.25, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.996016263961792, "rewards/meter/std": 0.00032200998975895345, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.996016263961792, "rewards/total_composite/std": 0.00032200998975895345, "reward": 0.996016263961792, "reward_std": 0.0003220097569283098, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015046549029648304, "sampling/sampling_logp_difference/max": 1.0829687118530273, "sampling/importance_sampling_ratio/min": 0.3910037875175476, "sampling/importance_sampling_ratio/mean": 1.0012880563735962, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06696707848459482, "clip_ratio/low_mean": 0.0028572361916303635, "clip_ratio/low_min": 0.0028572361916303635, "clip_ratio/high_mean": 0.011511339340358973, "clip_ratio/high_max": 0.011511339340358973, "clip_ratio/region_mean": 0.014368575531989336, "reward_total_mean": 0.996016263961792, "reward_meter_mean": 0.996016263961792, "reward_meter_std": 0.00032200998975895345, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.996016263961792, "reward_total_composite_std": 0.00032200998975895345} {"timestamp_utc": "2026-04-12T01:22:56Z", "mode": "train", "global_step": 2072, "epoch": 0.08322287825842471, "loss": 0.003, "grad_norm": 3.793785572052002, "learning_rate": 3.7242424242424246e-06, "num_tokens": 4663829.0, "completions/mean_length": 151.875, "completions/min_length": 149.0, "completions/max_length": 156.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 151.875, "completions/min_terminated_length": 149.0, "completions/max_terminated_length": 156.0, "rewards/meter/mean": 0.9847034215927124, "rewards/meter/std": 0.01465265266597271, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.8616158962249756, "rewards/total_composite/std": 0.05130418762564659, "reward": 0.8616158962249756, "reward_std": 0.05130419507622719, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02519957348704338, "sampling/sampling_logp_difference/max": 2.1398165225982666, "sampling/importance_sampling_ratio/min": 0.11767642945051193, "sampling/importance_sampling_ratio/mean": 1.002585530281067, "sampling/importance_sampling_ratio/max": 1.8727295398712158, "entropy": 0.11779375281184912, "clip_ratio/low_mean": 0.014854058739729226, "clip_ratio/low_min": 0.014854058739729226, "clip_ratio/high_mean": 0.003246753243729472, "clip_ratio/high_max": 0.003246753243729472, "clip_ratio/region_mean": 0.018100811983458698, "reward_total_mean": 0.8616158962249756, "reward_meter_mean": 0.9847034215927124, "reward_meter_std": 0.01465265266597271, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.8616158962249756, "reward_total_composite_std": 0.05130418762564659} {"timestamp_utc": "2026-04-12T01:23:01Z", "mode": "train", "global_step": 2073, "epoch": 0.08326304374020967, "loss": -0.0078, "grad_norm": 3.991828203201294, "learning_rate": 3.7212121212121215e-06, "num_tokens": 4665887.0, "completions/mean_length": 112.25, "completions/min_length": 108.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.25, "completions/min_terminated_length": 108.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.99208664894104, "rewards/meter/std": 0.012512140907347202, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.99208664894104, "rewards/total_composite/std": 0.012512140907347202, "reward": 0.99208664894104, "reward_std": 0.012512128800153732, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04193767532706261, "sampling/sampling_logp_difference/max": 1.4444141387939453, "sampling/importance_sampling_ratio/min": 0.23588424921035767, "sampling/importance_sampling_ratio/mean": 1.0028138160705566, "sampling/importance_sampling_ratio/max": 1.9161767959594727, "entropy": 0.3353299479931593, "clip_ratio/low_mean": 0.009154040599241853, "clip_ratio/low_min": 0.009154040599241853, "clip_ratio/high_mean": 0.023355348501354456, "clip_ratio/high_max": 0.023355348501354456, "clip_ratio/region_mean": 0.03250938910059631, "reward_total_mean": 0.99208664894104, "reward_meter_mean": 0.99208664894104, "reward_meter_std": 0.012512140907347202, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.99208664894104, "reward_total_composite_std": 0.012512140907347202} {"timestamp_utc": "2026-04-12T01:23:06Z", "mode": "train", "global_step": 2074, "epoch": 0.08330320922199462, "loss": 0.0016, "grad_norm": 1.4791640043258667, "learning_rate": 3.7181818181818187e-06, "num_tokens": 4667687.0, "completions/mean_length": 63.0, "completions/min_length": 63.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.0, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9969918131828308, "rewards/meter/std": 0.00010383435437688604, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9969918131828308, "rewards/total_composite/std": 0.00010383435437688604, "reward": 0.9969918131828308, "reward_std": 0.00010383195331087336, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005412998143583536, "sampling/sampling_logp_difference/max": 0.561959981918335, "sampling/importance_sampling_ratio/min": 0.7295204997062683, "sampling/importance_sampling_ratio/mean": 1.0029563903808594, "sampling/importance_sampling_ratio/max": 1.7541072368621826, "entropy": 0.02836236241273582, "clip_ratio/low_mean": 0.0019841270986944437, "clip_ratio/low_min": 0.0019841270986944437, "clip_ratio/high_mean": 0.0019841270986944437, "clip_ratio/high_max": 0.0019841270986944437, "clip_ratio/region_mean": 0.003968254197388887, "reward_total_mean": 0.9969918131828308, "reward_meter_mean": 0.9969918131828308, "reward_meter_std": 0.00010383435437688604, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9969918131828308, "reward_total_composite_std": 0.00010383435437688604} {"timestamp_utc": "2026-04-12T01:23:10Z", "mode": "train", "global_step": 2075, "epoch": 0.08334337470377957, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.7151515151515156e-06, "num_tokens": 4669527.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9990598559379578, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990598559379578, "rewards/total_composite/std": 0.0, "reward": 0.9990598559379578, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002672372793313116, "sampling/sampling_logp_difference/max": 0.008290700614452362, "sampling/importance_sampling_ratio/min": 0.9917435646057129, "sampling/importance_sampling_ratio/mean": 1.0002044439315796, "sampling/importance_sampling_ratio/max": 1.0042898654937744, "entropy": 0.0028630376618821174, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9990598559379578, "reward_meter_mean": 0.9990598559379578, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990598559379578, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:23:15Z", "mode": "train", "global_step": 2076, "epoch": 0.08338354018556453, "loss": -0.0008, "grad_norm": 2.183760404586792, "learning_rate": 3.7121212121212124e-06, "num_tokens": 4671885.0, "completions/mean_length": 116.75, "completions/min_length": 116.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.75, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.996731698513031, "rewards/meter/std": 3.716555511346087e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.8721410036087036, "rewards/total_composite/std": 0.050359878689050674, "reward": 0.8721410036087036, "reward_std": 0.05035989359021187, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008479885756969452, "sampling/sampling_logp_difference/max": 0.6767293214797974, "sampling/importance_sampling_ratio/min": 0.5082767009735107, "sampling/importance_sampling_ratio/mean": 1.000087022781372, "sampling/importance_sampling_ratio/max": 1.463352918624878, "entropy": 0.04457000829279423, "clip_ratio/low_mean": 0.005342192598618567, "clip_ratio/low_min": 0.005342192598618567, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.005342192598618567, "reward_total_mean": 0.8721410036087036, "reward_meter_mean": 0.996731698513031, "reward_meter_std": 3.716555511346087e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.8721410036087036, "reward_total_composite_std": 0.050359878689050674} {"timestamp_utc": "2026-04-12T01:23:20Z", "mode": "train", "global_step": 2077, "epoch": 0.08342370566734948, "loss": -0.005, "grad_norm": 5.309196949005127, "learning_rate": 3.7090909090909092e-06, "num_tokens": 4673787.0, "completions/mean_length": 68.75, "completions/min_length": 67.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.75, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9131183624267578, "rewards/meter/std": 0.07230029255151749, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9131183624267578, "rewards/total_composite/std": 0.07230029255151749, "reward": 0.9131183624267578, "reward_std": 0.07230029255151749, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007340815383940935, "sampling/sampling_logp_difference/max": 0.4846305847167969, "sampling/importance_sampling_ratio/min": 0.6702300310134888, "sampling/importance_sampling_ratio/mean": 1.0044362545013428, "sampling/importance_sampling_ratio/max": 1.6235750913619995, "entropy": 0.05782870203256607, "clip_ratio/low_mean": 0.0037313431967049837, "clip_ratio/low_min": 0.0037313431967049837, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0037313431967049837, "reward_total_mean": 0.9131183624267578, "reward_meter_mean": 0.9131183624267578, "reward_meter_std": 0.07230029255151749, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9131183624267578, "reward_total_composite_std": 0.07230029255151749} {"timestamp_utc": "2026-04-12T01:23:26Z", "mode": "train", "global_step": 2078, "epoch": 0.08346387114913444, "loss": -0.0024, "grad_norm": 2.607077121734619, "learning_rate": 3.7060606060606065e-06, "num_tokens": 4676302.0, "completions/mean_length": 148.375, "completions/min_length": 146.0, "completions/max_length": 153.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 148.375, "completions/min_terminated_length": 146.0, "completions/max_terminated_length": 153.0, "rewards/meter/mean": 0.9990102648735046, "rewards/meter/std": 0.00021416397066786885, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.910714328289032, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.909817099571228, "rewards/total_composite/std": 0.0739244893193245, "reward": 0.909817099571228, "reward_std": 0.0739244818687439, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023411214351654053, "sampling/sampling_logp_difference/max": 1.31553316116333, "sampling/importance_sampling_ratio/min": 0.26833122968673706, "sampling/importance_sampling_ratio/mean": 1.0051114559173584, "sampling/importance_sampling_ratio/max": 1.6944489479064941, "entropy": 0.20128737576305866, "clip_ratio/low_mean": 0.006791699095629156, "clip_ratio/low_min": 0.006791699095629156, "clip_ratio/high_mean": 0.00663501548115164, "clip_ratio/high_max": 0.00663501548115164, "clip_ratio/region_mean": 0.013426714576780796, "reward_total_mean": 0.909817099571228, "reward_meter_mean": 0.9990102648735046, "reward_meter_std": 0.00021416397066786885, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.910714328289032, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.909817099571228, "reward_total_composite_std": 0.0739244893193245} {"timestamp_utc": "2026-04-12T01:23:30Z", "mode": "train", "global_step": 2079, "epoch": 0.08350403663091939, "loss": -0.0002, "grad_norm": 0.3224610984325409, "learning_rate": 3.7030303030303033e-06, "num_tokens": 4678159.0, "completions/mean_length": 66.125, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.125, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9980389475822449, "rewards/meter/std": 2.6020999939646572e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980389475822449, "rewards/total_composite/std": 2.6020999939646572e-05, "reward": 0.9980389475822449, "reward_std": 2.602726817713119e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007125611882656813, "sampling/sampling_logp_difference/max": 1.095078468322754, "sampling/importance_sampling_ratio/min": 0.3345133364200592, "sampling/importance_sampling_ratio/mean": 0.9994921088218689, "sampling/importance_sampling_ratio/max": 1.2167119979858398, "entropy": 0.026706934673711658, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/high_mean": 0.0037313431967049837, "clip_ratio/high_max": 0.0037313431967049837, "clip_ratio/region_mean": 0.00562528264708817, "reward_total_mean": 0.9980389475822449, "reward_meter_mean": 0.9980389475822449, "reward_meter_std": 2.6020999939646572e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9980389475822449, "reward_total_composite_std": 2.6020999939646572e-05} {"timestamp_utc": "2026-04-12T01:23:35Z", "mode": "train", "global_step": 2080, "epoch": 0.08354420211270434, "loss": -0.0081, "grad_norm": 9.717095375061035, "learning_rate": 3.7e-06, "num_tokens": 4679892.0, "completions/mean_length": 39.625, "completions/min_length": 38.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.625, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.9970946311950684, "rewards/meter/std": 0.003818577155470848, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9970946311950684, "rewards/total_composite/std": 0.003818577155470848, "reward": 0.9970946311950684, "reward_std": 0.0038185729645192623, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.058814503252506256, "sampling/sampling_logp_difference/max": 1.591008186340332, "sampling/importance_sampling_ratio/min": 0.2037201076745987, "sampling/importance_sampling_ratio/mean": 0.9983224868774414, "sampling/importance_sampling_ratio/max": 1.6322083473205566, "entropy": 0.395177373662591, "clip_ratio/low_mean": 0.02227564202621579, "clip_ratio/low_min": 0.02227564202621579, "clip_ratio/high_mean": 0.03190367738716304, "clip_ratio/high_max": 0.03190367738716304, "clip_ratio/region_mean": 0.054179319413378835, "reward_total_mean": 0.9970946311950684, "reward_meter_mean": 0.9970946311950684, "reward_meter_std": 0.003818577155470848, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9970946311950684, "reward_total_composite_std": 0.003818577155470848} {"timestamp_utc": "2026-04-12T01:23:39Z", "mode": "train", "global_step": 2081, "epoch": 0.0835843675944893, "loss": -0.0053, "grad_norm": 7.125683784484863, "learning_rate": 3.6969696969696974e-06, "num_tokens": 4681889.0, "completions/mean_length": 75.625, "completions/min_length": 74.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.625, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.8755629062652588, "rewards/meter/std": 0.3493680953979492, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8755629062652588, "rewards/total_composite/std": 0.3493680953979492, "reward": 0.8755629062652588, "reward_std": 0.3493681252002716, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.032194141298532486, "sampling/sampling_logp_difference/max": 0.9960083961486816, "sampling/importance_sampling_ratio/min": 0.3693508207798004, "sampling/importance_sampling_ratio/mean": 1.0054106712341309, "sampling/importance_sampling_ratio/max": 1.5550665855407715, "entropy": 0.22988875024020672, "clip_ratio/low_mean": 0.00844594556838274, "clip_ratio/low_min": 0.00844594556838274, "clip_ratio/high_mean": 0.009914012742228806, "clip_ratio/high_max": 0.009914012742228806, "clip_ratio/region_mean": 0.018359958310611546, "reward_total_mean": 0.8755629062652588, "reward_meter_mean": 0.8755629062652588, "reward_meter_std": 0.3493680953979492, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8755629062652588, "reward_total_composite_std": 0.3493680953979492} {"timestamp_utc": "2026-04-12T01:23:44Z", "mode": "train", "global_step": 2082, "epoch": 0.08362453307627425, "loss": -0.0013, "grad_norm": 0.7917746305465698, "learning_rate": 3.6939393939393942e-06, "num_tokens": 4683848.0, "completions/mean_length": 65.875, "completions/min_length": 65.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.875, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9979181289672852, "rewards/meter/std": 0.00028089084662497044, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979181289672852, "rewards/total_composite/std": 0.00028089084662497044, "reward": 0.9979181289672852, "reward_std": 0.00028088997351005673, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008248316124081612, "sampling/sampling_logp_difference/max": 1.082160472869873, "sampling/importance_sampling_ratio/min": 0.3388626277446747, "sampling/importance_sampling_ratio/mean": 0.9990794658660889, "sampling/importance_sampling_ratio/max": 1.159103274345398, "entropy": 0.026136466301977634, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.001923076924867928, "clip_ratio/high_max": 0.001923076924867928, "clip_ratio/region_mean": 0.001923076924867928, "reward_total_mean": 0.9979181289672852, "reward_meter_mean": 0.9979181289672852, "reward_meter_std": 0.00028089084662497044, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9979181289672852, "reward_total_composite_std": 0.00028089084662497044} {"timestamp_utc": "2026-04-12T01:23:49Z", "mode": "train", "global_step": 2083, "epoch": 0.0836646985580592, "loss": 0.0021, "grad_norm": 3.155012845993042, "learning_rate": 3.690909090909091e-06, "num_tokens": 4686452.0, "completions/mean_length": 147.5, "completions/min_length": 144.0, "completions/max_length": 150.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 147.5, "completions/min_terminated_length": 144.0, "completions/max_terminated_length": 150.0, "rewards/meter/mean": 0.9979666471481323, "rewards/meter/std": 0.0005141000729054213, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9801425337791443, "rewards/total_composite/std": 0.05034194886684418, "reward": 0.9801425337791443, "reward_std": 0.050341930240392685, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.039336614310741425, "sampling/sampling_logp_difference/max": 1.2754813432693481, "sampling/importance_sampling_ratio/min": 0.27929648756980896, "sampling/importance_sampling_ratio/mean": 1.0034173727035522, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2962766755372286, "clip_ratio/low_mean": 0.0051369862630963326, "clip_ratio/low_min": 0.0051369862630963326, "clip_ratio/high_mean": 0.03633964783512056, "clip_ratio/high_max": 0.03633964783512056, "clip_ratio/region_mean": 0.04147663409821689, "reward_total_mean": 0.9801425337791443, "reward_meter_mean": 0.9979666471481323, "reward_meter_std": 0.0005141000729054213, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.9801425337791443, "reward_total_composite_std": 0.05034194886684418} {"timestamp_utc": "2026-04-12T01:23:55Z", "mode": "train", "global_step": 2084, "epoch": 0.08370486403984416, "loss": 0.0021, "grad_norm": 5.227519989013672, "learning_rate": 3.687878787878788e-06, "num_tokens": 4689035.0, "completions/mean_length": 151.875, "completions/min_length": 142.0, "completions/max_length": 160.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 151.875, "completions/min_terminated_length": 142.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.8921954035758972, "rewards/meter/std": 0.1858879029750824, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8928571939468384, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.7994784116744995, "rewards/total_composite/std": 0.18709427118301392, "reward": 0.7994784116744995, "reward_std": 0.18709427118301392, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031672801822423935, "sampling/sampling_logp_difference/max": 2.7413718700408936, "sampling/importance_sampling_ratio/min": 0.06448182463645935, "sampling/importance_sampling_ratio/mean": 1.0021029710769653, "sampling/importance_sampling_ratio/max": 1.7797260284423828, "entropy": 0.14115788787603378, "clip_ratio/low_mean": 0.004023639601655304, "clip_ratio/low_min": 0.004023639601655304, "clip_ratio/high_mean": 0.020977655542083085, "clip_ratio/high_max": 0.020977655542083085, "clip_ratio/region_mean": 0.02500129514373839, "reward_total_mean": 0.7994784116744995, "reward_meter_mean": 0.8921954035758972, "reward_meter_std": 0.1858879029750824, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8928571939468384, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.7994784116744995, "reward_total_composite_std": 0.18709427118301392} {"timestamp_utc": "2026-04-12T01:24:03Z", "mode": "train", "global_step": 2085, "epoch": 0.08374502952162911, "loss": 0.0076, "grad_norm": 1.964224934577942, "learning_rate": 3.684848484848485e-06, "num_tokens": 4693004.0, "completions/mean_length": 294.125, "completions/min_length": 288.0, "completions/max_length": 310.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 294.125, "completions/min_terminated_length": 288.0, "completions/max_terminated_length": 310.0, "rewards/meter/mean": 0.998891294002533, "rewards/meter/std": 0.0004529697762336582, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8760416507720947, "rewards/repeat_penalty/std": 0.023332269862294197, "rewards/total_composite/mean": 0.8750678896903992, "rewards/total_composite/std": 0.02319573611021042, "reward": 0.8750678896903992, "reward_std": 0.023195745423436165, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025770394131541252, "sampling/sampling_logp_difference/max": 2.1700761318206787, "sampling/importance_sampling_ratio/min": 0.1141689196228981, "sampling/importance_sampling_ratio/mean": 1.0067555904388428, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2046606820076704, "clip_ratio/low_mean": 0.018270427826792, "clip_ratio/low_min": 0.018270427826792, "clip_ratio/high_mean": 0.003448275849223137, "clip_ratio/high_max": 0.003448275849223137, "clip_ratio/region_mean": 0.02171870367601514, "reward_total_mean": 0.8750678896903992, "reward_meter_mean": 0.998891294002533, "reward_meter_std": 0.0004529697762336582, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8760416507720947, "reward_repeat_penalty_std": 0.023332269862294197, "reward_total_composite_mean": 0.8750678896903992, "reward_total_composite_std": 0.02319573611021042} {"timestamp_utc": "2026-04-12T01:24:10Z", "mode": "train", "global_step": 2086, "epoch": 0.08378519500341407, "loss": -0.0218, "grad_norm": 1.5824189186096191, "learning_rate": 3.681818181818182e-06, "num_tokens": 4696909.0, "completions/mean_length": 291.125, "completions/min_length": 270.0, "completions/max_length": 306.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 291.125, "completions/min_terminated_length": 270.0, "completions/max_terminated_length": 306.0, "rewards/meter/mean": 0.9986486434936523, "rewards/meter/std": 0.0003730578755494207, "rewards/count_adherence/mean": 0.984375, "rewards/count_adherence/std": 0.04419417306780815, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8737351298332214, "rewards/repeat_penalty/std": 0.06906846165657043, "rewards/total_composite/mean": 0.8602873086929321, "rewards/total_composite/std": 0.09172172844409943, "reward": 0.8602873086929321, "reward_std": 0.09172171354293823, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0357806496322155, "sampling/sampling_logp_difference/max": 15.902763366699219, "sampling/importance_sampling_ratio/min": 1.240273945768422e-07, "sampling/importance_sampling_ratio/mean": 1.003360390663147, "sampling/importance_sampling_ratio/max": 1.9022753238677979, "entropy": 0.23322945460677147, "clip_ratio/low_mean": 0.007962123490869999, "clip_ratio/low_min": 0.007962123490869999, "clip_ratio/high_mean": 0.020660461392253637, "clip_ratio/high_max": 0.020660461392253637, "clip_ratio/region_mean": 0.028622584883123636, "reward_total_mean": 0.8602873086929321, "reward_meter_mean": 0.9986486434936523, "reward_meter_std": 0.0003730578755494207, "reward_count_adherence_mean": 0.984375, "reward_count_adherence_std": 0.04419417306780815, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8737351298332214, "reward_repeat_penalty_std": 0.06906846165657043, "reward_total_composite_mean": 0.8602873086929321, "reward_total_composite_std": 0.09172172844409943} {"timestamp_utc": "2026-04-12T01:24:14Z", "mode": "train", "global_step": 2087, "epoch": 0.08382536048519902, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.678787878787879e-06, "num_tokens": 4698765.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9990598559379578, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990598559379578, "rewards/total_composite/std": 0.0, "reward": 0.9990598559379578, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.000278584222542122, "sampling/sampling_logp_difference/max": 0.006727563217282295, "sampling/importance_sampling_ratio/min": 0.9977940917015076, "sampling/importance_sampling_ratio/mean": 1.0002596378326416, "sampling/importance_sampling_ratio/max": 1.0067503452301025, "entropy": 0.0027300505607854575, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9990598559379578, "reward_meter_mean": 0.9990598559379578, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990598559379578, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:24:20Z", "mode": "train", "global_step": 2088, "epoch": 0.08386552596698398, "loss": -0.0178, "grad_norm": 3.470649480819702, "learning_rate": 3.6757575757575757e-06, "num_tokens": 4701006.0, "completions/mean_length": 112.125, "completions/min_length": 102.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.125, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.9967917203903198, "rewards/meter/std": 0.002291632117703557, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9967917203903198, "rewards/total_composite/std": 0.002291632117703557, "reward": 0.9967917203903198, "reward_std": 0.002291641663759947, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02775193564593792, "sampling/sampling_logp_difference/max": 1.1847543716430664, "sampling/importance_sampling_ratio/min": 0.3058212697505951, "sampling/importance_sampling_ratio/mean": 1.0074383020401, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.18240004498511553, "clip_ratio/low_mean": 0.007895297603681684, "clip_ratio/low_min": 0.007895297603681684, "clip_ratio/high_mean": 0.013125053141266108, "clip_ratio/high_max": 0.013125053141266108, "clip_ratio/region_mean": 0.02102035074494779, "reward_total_mean": 0.9967917203903198, "reward_meter_mean": 0.9967917203903198, "reward_meter_std": 0.002291632117703557, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9967917203903198, "reward_total_composite_std": 0.002291632117703557} {"timestamp_utc": "2026-04-12T01:24:24Z", "mode": "train", "global_step": 2089, "epoch": 0.08390569144876893, "loss": -0.0005, "grad_norm": 1.3091340065002441, "learning_rate": 3.672727272727273e-06, "num_tokens": 4702766.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9990187883377075, "rewards/meter/std": 0.00011603027814999223, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990187883377075, "rewards/total_composite/std": 0.00011603027814999223, "reward": 0.9990187883377075, "reward_std": 0.00011603629536693916, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0023684005718678236, "sampling/sampling_logp_difference/max": 1.0964341163635254, "sampling/importance_sampling_ratio/min": 0.3340601921081543, "sampling/importance_sampling_ratio/mean": 0.9990017414093018, "sampling/importance_sampling_ratio/max": 1.0052270889282227, "entropy": 0.0029864944226574153, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9990187883377075, "reward_meter_mean": 0.9990187883377075, "reward_meter_std": 0.00011603027814999223, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990187883377075, "reward_total_composite_std": 0.00011603027814999223} {"timestamp_utc": "2026-04-12T01:24:29Z", "mode": "train", "global_step": 2090, "epoch": 0.08394585693055388, "loss": -0.0218, "grad_norm": 2.0700643062591553, "learning_rate": 3.6696969696969697e-06, "num_tokens": 4704838.0, "completions/mean_length": 86.0, "completions/min_length": 80.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.0, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.9953513741493225, "rewards/meter/std": 0.0020426749251782894, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9953513741493225, "rewards/total_composite/std": 0.0020426749251782894, "reward": 0.9953513741493225, "reward_std": 0.0020426807459443808, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013093557208776474, "sampling/sampling_logp_difference/max": 1.2429170608520508, "sampling/importance_sampling_ratio/min": 0.288541316986084, "sampling/importance_sampling_ratio/mean": 1.0013517141342163, "sampling/importance_sampling_ratio/max": 1.8694008588790894, "entropy": 0.05126679269596934, "clip_ratio/low_mean": 0.0015625000232830644, "clip_ratio/low_min": 0.0015625000232830644, "clip_ratio/high_mean": 0.0100741779897362, "clip_ratio/high_max": 0.0100741779897362, "clip_ratio/region_mean": 0.011636678013019264, "reward_total_mean": 0.9953513741493225, "reward_meter_mean": 0.9953513741493225, "reward_meter_std": 0.0020426749251782894, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9953513741493225, "reward_total_composite_std": 0.0020426749251782894} {"timestamp_utc": "2026-04-12T01:24:34Z", "mode": "train", "global_step": 2091, "epoch": 0.08398602241233884, "loss": 0.0105, "grad_norm": 2.3087821006774902, "learning_rate": 3.6666666666666666e-06, "num_tokens": 4707486.0, "completions/mean_length": 147.0, "completions/min_length": 144.0, "completions/max_length": 150.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 147.0, "completions/min_terminated_length": 144.0, "completions/max_terminated_length": 150.0, "rewards/meter/mean": 0.9949081540107727, "rewards/meter/std": 0.004385494627058506, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.959259033203125, "rewards/total_composite/std": 0.0641113817691803, "reward": 0.959259033203125, "reward_std": 0.0641113668680191, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027278585359454155, "sampling/sampling_logp_difference/max": 1.1111698150634766, "sampling/importance_sampling_ratio/min": 0.3291736841201782, "sampling/importance_sampling_ratio/mean": 1.0080565214157104, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.19478315208107233, "clip_ratio/low_mean": 0.0033445945591665804, "clip_ratio/low_min": 0.0033445945591665804, "clip_ratio/high_mean": 0.017056229757145047, "clip_ratio/high_max": 0.017056229757145047, "clip_ratio/region_mean": 0.020400824316311628, "reward_total_mean": 0.959259033203125, "reward_meter_mean": 0.9949081540107727, "reward_meter_std": 0.004385494627058506, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.959259033203125, "reward_total_composite_std": 0.0641113817691803} {"timestamp_utc": "2026-04-12T01:24:39Z", "mode": "train", "global_step": 2092, "epoch": 0.08402618789412379, "loss": 0.0061, "grad_norm": 6.707172393798828, "learning_rate": 3.6636363636363643e-06, "num_tokens": 4709052.0, "completions/mean_length": 38.75, "completions/min_length": 37.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.75, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.9994734525680542, "rewards/meter/std": 0.0001241376594407484, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994734525680542, "rewards/total_composite/std": 0.0001241376594407484, "reward": 0.9994734525680542, "reward_std": 0.00012412367505021393, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031195173040032387, "sampling/sampling_logp_difference/max": 0.9616420269012451, "sampling/importance_sampling_ratio/min": 0.3822647035121918, "sampling/importance_sampling_ratio/mean": 0.9975427389144897, "sampling/importance_sampling_ratio/max": 1.5661958456039429, "entropy": 0.1498966608196497, "clip_ratio/low_mean": 0.0033783784601837397, "clip_ratio/low_min": 0.0033783784601837397, "clip_ratio/high_mean": 0.012833506800234318, "clip_ratio/high_max": 0.012833506800234318, "clip_ratio/region_mean": 0.016211885260418057, "reward_total_mean": 0.9994734525680542, "reward_meter_mean": 0.9994734525680542, "reward_meter_std": 0.0001241376594407484, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994734525680542, "reward_total_composite_std": 0.0001241376594407484} {"timestamp_utc": "2026-04-12T01:24:43Z", "mode": "train", "global_step": 2093, "epoch": 0.08406635337590875, "loss": 0.0237, "grad_norm": 7.211493015289307, "learning_rate": 3.660606060606061e-06, "num_tokens": 4710458.0, "completions/mean_length": 37.75, "completions/min_length": 37.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.75, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.9790675640106201, "rewards/meter/std": 0.02823478914797306, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9790675640106201, "rewards/total_composite/std": 0.02823478914797306, "reward": 0.9790675640106201, "reward_std": 0.02823478914797306, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014203748665750027, "sampling/sampling_logp_difference/max": 0.29107916355133057, "sampling/importance_sampling_ratio/min": 0.7474565505981445, "sampling/importance_sampling_ratio/mean": 1.0054656267166138, "sampling/importance_sampling_ratio/max": 1.3320231437683105, "entropy": 0.12450700253248215, "clip_ratio/low_mean": 0.009615384740754962, "clip_ratio/low_min": 0.009615384740754962, "clip_ratio/high_mean": 0.006756756920367479, "clip_ratio/high_max": 0.006756756920367479, "clip_ratio/region_mean": 0.01637214166112244, "reward_total_mean": 0.9790675640106201, "reward_meter_mean": 0.9790675640106201, "reward_meter_std": 0.02823478914797306, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9790675640106201, "reward_total_composite_std": 0.02823478914797306} {"timestamp_utc": "2026-04-12T01:24:47Z", "mode": "train", "global_step": 2094, "epoch": 0.0841065188576937, "loss": 0.0017, "grad_norm": 0.3633219003677368, "learning_rate": 3.657575757575758e-06, "num_tokens": 4712186.0, "completions/mean_length": 69.0, "completions/min_length": 69.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.0, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.938814103603363, "rewards/meter/std": 0.0002475241490174085, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.938814103603363, "rewards/total_composite/std": 0.0002475241490174085, "reward": 0.938814103603363, "reward_std": 0.0002475333458278328, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0060532353818416595, "sampling/sampling_logp_difference/max": 0.7302188873291016, "sampling/importance_sampling_ratio/min": 0.4818035662174225, "sampling/importance_sampling_ratio/mean": 0.999586820602417, "sampling/importance_sampling_ratio/max": 1.0647722482681274, "entropy": 0.04171428643167019, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0036231884732842445, "clip_ratio/high_max": 0.0036231884732842445, "clip_ratio/region_mean": 0.0036231884732842445, "reward_total_mean": 0.938814103603363, "reward_meter_mean": 0.938814103603363, "reward_meter_std": 0.0002475241490174085, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.938814103603363, "reward_total_composite_std": 0.0002475241490174085} {"timestamp_utc": "2026-04-12T01:24:52Z", "mode": "train", "global_step": 2095, "epoch": 0.08414668433947865, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.654545454545455e-06, "num_tokens": 4714042.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9990598559379578, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990598559379578, "rewards/total_composite/std": 0.0, "reward": 0.9990598559379578, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0005561705911532044, "sampling/sampling_logp_difference/max": 0.016185477375984192, "sampling/importance_sampling_ratio/min": 0.9876941442489624, "sampling/importance_sampling_ratio/mean": 1.0003678798675537, "sampling/importance_sampling_ratio/max": 1.0163171291351318, "entropy": 0.00596828549169004, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9990598559379578, "reward_meter_mean": 0.9990598559379578, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990598559379578, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:24:56Z", "mode": "train", "global_step": 2096, "epoch": 0.0841868498212636, "loss": 0.006, "grad_norm": 3.099059581756592, "learning_rate": 3.651515151515152e-06, "num_tokens": 4716098.0, "completions/mean_length": 69.0, "completions/min_length": 68.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.0, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9360407590866089, "rewards/meter/std": 0.005241453181952238, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9360407590866089, "rewards/total_composite/std": 0.005241453181952238, "reward": 0.9360407590866089, "reward_std": 0.0052414704114198685, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011832116171717644, "sampling/sampling_logp_difference/max": 0.7093310356140137, "sampling/importance_sampling_ratio/min": 0.49197322130203247, "sampling/importance_sampling_ratio/mean": 1.001249074935913, "sampling/importance_sampling_ratio/max": 1.8257942199707031, "entropy": 0.07018918683752418, "clip_ratio/low_mean": 0.0018115942366421223, "clip_ratio/low_min": 0.0018115942366421223, "clip_ratio/high_mean": 0.0036231884732842445, "clip_ratio/high_max": 0.0036231884732842445, "clip_ratio/region_mean": 0.005434782709926367, "reward_total_mean": 0.9360407590866089, "reward_meter_mean": 0.9360407590866089, "reward_meter_std": 0.005241453181952238, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9360407590866089, "reward_total_composite_std": 0.005241453181952238} {"timestamp_utc": "2026-04-12T01:25:00Z", "mode": "train", "global_step": 2097, "epoch": 0.08422701530304856, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.648484848484849e-06, "num_tokens": 4717538.0, "completions/mean_length": 28.0, "completions/min_length": 28.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.0, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9942078590393066, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9942078590393066, "rewards/total_composite/std": 0.0, "reward": 0.9942078590393066, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0031366986222565174, "sampling/sampling_logp_difference/max": 0.15313977003097534, "sampling/importance_sampling_ratio/min": 0.8580098152160645, "sampling/importance_sampling_ratio/mean": 1.0009886026382446, "sampling/importance_sampling_ratio/max": 1.0437800884246826, "entropy": 0.030742917326278985, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9942078590393066, "reward_meter_mean": 0.9942078590393066, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9942078590393066, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:25:05Z", "mode": "train", "global_step": 2098, "epoch": 0.08426718078483351, "loss": 0.0272, "grad_norm": 2.8510286808013916, "learning_rate": 3.645454545454546e-06, "num_tokens": 4719239.0, "completions/mean_length": 75.625, "completions/min_length": 71.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.625, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9827600717544556, "rewards/meter/std": 0.0294841006398201, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9827600717544556, "rewards/total_composite/std": 0.0294841006398201, "reward": 0.9827600717544556, "reward_std": 0.029484104365110397, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.033739130944013596, "sampling/sampling_logp_difference/max": 1.151628017425537, "sampling/importance_sampling_ratio/min": 0.3161216974258423, "sampling/importance_sampling_ratio/mean": 0.9956676363945007, "sampling/importance_sampling_ratio/max": 1.6294512748718262, "entropy": 0.16028152592480183, "clip_ratio/low_mean": 0.0015625000232830644, "clip_ratio/low_min": 0.0015625000232830644, "clip_ratio/high_mean": 0.028136852895841002, "clip_ratio/high_max": 0.028136852895841002, "clip_ratio/region_mean": 0.029699352919124067, "reward_total_mean": 0.9827600717544556, "reward_meter_mean": 0.9827600717544556, "reward_meter_std": 0.0294841006398201, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9827600717544556, "reward_total_composite_std": 0.0294841006398201} {"timestamp_utc": "2026-04-12T01:25:10Z", "mode": "train", "global_step": 2099, "epoch": 0.08430734626661847, "loss": -0.0241, "grad_norm": 4.330075740814209, "learning_rate": 3.642424242424243e-06, "num_tokens": 4721395.0, "completions/mean_length": 105.5, "completions/min_length": 96.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.5, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.8654639720916748, "rewards/meter/std": 0.23051653802394867, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8491353988647461, "rewards/total_composite/std": 0.25138652324676514, "reward": 0.8491353988647461, "reward_std": 0.25138649344444275, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03153692185878754, "sampling/sampling_logp_difference/max": 1.3270950317382812, "sampling/importance_sampling_ratio/min": 0.3179101049900055, "sampling/importance_sampling_ratio/mean": 1.007623553276062, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.29543587751686573, "clip_ratio/low_mean": 0.008760618977248669, "clip_ratio/low_min": 0.008760618977248669, "clip_ratio/high_mean": 0.016345139825716615, "clip_ratio/high_max": 0.016345139825716615, "clip_ratio/region_mean": 0.025105758802965283, "reward_total_mean": 0.8491353988647461, "reward_meter_mean": 0.8654639720916748, "reward_meter_std": 0.23051653802394867, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.8491353988647461, "reward_total_composite_std": 0.25138652324676514} {"timestamp_utc": "2026-04-12T01:25:15Z", "mode": "train", "global_step": 2100, "epoch": 0.08434751174840342, "loss": 0.0052, "grad_norm": 7.931367874145508, "learning_rate": 3.6393939393939398e-06, "num_tokens": 4723535.0, "completions/mean_length": 86.5, "completions/min_length": 83.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.5, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.9882360100746155, "rewards/meter/std": 0.022600755095481873, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9882360100746155, "rewards/total_composite/std": 0.022600755095481873, "reward": 0.9882360100746155, "reward_std": 0.022600755095481873, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022525055333971977, "sampling/sampling_logp_difference/max": 1.4979562759399414, "sampling/importance_sampling_ratio/min": 0.22358666360378265, "sampling/importance_sampling_ratio/mean": 1.0013149976730347, "sampling/importance_sampling_ratio/max": 1.8274431228637695, "entropy": 0.20394995296373963, "clip_ratio/low_mean": 0.0060240961611270905, "clip_ratio/low_min": 0.0060240961611270905, "clip_ratio/high_mean": 0.008620689623057842, "clip_ratio/high_max": 0.008620689623057842, "clip_ratio/region_mean": 0.014644785784184933, "reward_total_mean": 0.9882360100746155, "reward_meter_mean": 0.9882360100746155, "reward_meter_std": 0.022600755095481873, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9882360100746155, "reward_total_composite_std": 0.022600755095481873} {"timestamp_utc": "2026-04-12T01:26:24Z", "mode": "eval", "global_step": 2100, "epoch": 0.08434751174840342, "eval_loss": NaN, "eval_runtime": 68.5013, "eval_samples_per_second": 1.518, "eval_steps_per_second": 0.19, "eval_num_tokens": 4723535.0, "eval_completions/mean_length": 201.65384615384616, "eval_completions/min_length": 61.92307692307692, "eval_completions/max_length": 356.53846153846155, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 201.65384615384616, "eval_completions/min_terminated_length": 61.92307692307692, "eval_completions/max_terminated_length": 356.53846153846155, "eval_rewards/meter/mean": 0.7131731968659621, "eval_rewards/meter/std": 0.4128540616769057, "eval_rewards/count_adherence/mean": 0.9138026375036973, "eval_rewards/count_adherence/std": 0.12064289301633835, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/repeat_penalty/mean": 0.8164694079985986, "eval_rewards/repeat_penalty/std": 0.1486260168827497, "eval_rewards/total_composite/mean": 0.542255353469115, "eval_rewards/total_composite/std": 0.3623279837461618, "eval_reward": 0.542255353469115, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.013091834047092842, "eval_sampling/sampling_logp_difference/max": 0.9667405440257146, "eval_sampling/importance_sampling_ratio/min": 0.38750078357183015, "eval_sampling/importance_sampling_ratio/mean": 1.0036665751383855, "eval_sampling/importance_sampling_ratio/max": 1.501889733167795, "eval_entropy": 0.12711333655394041, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.542255353469115, "eval_reward_meter_mean": 0.7131731968659621, "eval_reward_meter_std": 0.4128540616769057, "eval_reward_count_adherence_mean": 0.9138026375036973, "eval_reward_count_adherence_std": 0.12064289301633835, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_repeat_penalty_mean": 0.8164694079985986, "eval_reward_repeat_penalty_std": 0.1486260168827497, "eval_reward_total_composite_mean": 0.542255353469115, "eval_reward_total_composite_std": 0.3623279837461618} {"timestamp_utc": "2026-04-12T01:26:33Z", "mode": "train", "global_step": 2101, "epoch": 0.08438767723018838, "loss": -0.0478, "grad_norm": 1.2342872619628906, "learning_rate": 3.6363636363636366e-06, "num_tokens": 4726322.0, "completions/mean_length": 176.375, "completions/min_length": 160.0, "completions/max_length": 192.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 176.375, "completions/min_terminated_length": 160.0, "completions/max_terminated_length": 192.0, "rewards/meter/mean": 0.9766891598701477, "rewards/meter/std": 0.022879159078001976, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.1035098284482956, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.761904776096344, "rewards/repeat_penalty/std": 0.07493293285369873, "rewards/total_composite/mean": 0.6860010623931885, "rewards/total_composite/std": 0.08363299816846848, "reward": 0.6860010623931885, "reward_std": 0.08363301306962967, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01796703413128853, "sampling/sampling_logp_difference/max": 1.0117740631103516, "sampling/importance_sampling_ratio/min": 0.36357343196868896, "sampling/importance_sampling_ratio/mean": 1.0021946430206299, "sampling/importance_sampling_ratio/max": 1.9307383298873901, "entropy": 0.11573350336402655, "clip_ratio/low_mean": 0.007700952875893563, "clip_ratio/low_min": 0.007700952875893563, "clip_ratio/high_mean": 0.00595723238075152, "clip_ratio/high_max": 0.00595723238075152, "clip_ratio/region_mean": 0.013658185256645083, "reward_total_mean": 0.6860010623931885, "reward_meter_mean": 0.9766891598701477, "reward_meter_std": 0.022879159078001976, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.1035098284482956, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.761904776096344, "reward_repeat_penalty_std": 0.07493293285369873, "reward_total_composite_mean": 0.6860010623931885, "reward_total_composite_std": 0.08363299816846848} {"timestamp_utc": "2026-04-12T01:26:37Z", "mode": "train", "global_step": 2102, "epoch": 0.08442784271197333, "loss": 0.0086, "grad_norm": 14.514649391174316, "learning_rate": 3.633333333333334e-06, "num_tokens": 4727945.0, "completions/mean_length": 28.875, "completions/min_length": 28.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.875, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9928563833236694, "rewards/meter/std": 0.0002995376707985997, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9928563833236694, "rewards/total_composite/std": 0.0002995376707985997, "reward": 0.9928563833236694, "reward_std": 0.0002995316463056952, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0030352873727679253, "sampling/sampling_logp_difference/max": 0.14670616388320923, "sampling/importance_sampling_ratio/min": 0.8847241401672363, "sampling/importance_sampling_ratio/mean": 1.0009772777557373, "sampling/importance_sampling_ratio/max": 1.1580135822296143, "entropy": 0.03053366276435554, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9928563833236694, "reward_meter_mean": 0.9928563833236694, "reward_meter_std": 0.0002995376707985997, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9928563833236694, "reward_total_composite_std": 0.0002995376707985997} {"timestamp_utc": "2026-04-12T01:26:45Z", "mode": "train", "global_step": 2103, "epoch": 0.08446800819375828, "loss": -0.0264, "grad_norm": 1.2050065994262695, "learning_rate": 3.6303030303030307e-06, "num_tokens": 4731675.0, "completions/mean_length": 279.25, "completions/min_length": 257.0, "completions/max_length": 286.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 279.25, "completions/min_terminated_length": 257.0, "completions/max_terminated_length": 286.0, "rewards/meter/mean": 0.9298574924468994, "rewards/meter/std": 0.1897430717945099, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8583333492279053, "rewards/repeat_penalty/std": 0.19002924859523773, "rewards/total_composite/mean": 0.8288908004760742, "rewards/total_composite/std": 0.26403236389160156, "reward": 0.8288908004760742, "reward_std": 0.2640323340892792, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023323198780417442, "sampling/sampling_logp_difference/max": 1.9472055435180664, "sampling/importance_sampling_ratio/min": 0.14267219603061676, "sampling/importance_sampling_ratio/mean": 1.0039503574371338, "sampling/importance_sampling_ratio/max": 1.7007434368133545, "entropy": 0.272721977904439, "clip_ratio/low_mean": 0.00048638132284395397, "clip_ratio/low_min": 0.00048638132284395397, "clip_ratio/high_mean": 0.02126871724613011, "clip_ratio/high_max": 0.02126871724613011, "clip_ratio/region_mean": 0.021755098568974063, "reward_total_mean": 0.8288908004760742, "reward_meter_mean": 0.9298574924468994, "reward_meter_std": 0.1897430717945099, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8583333492279053, "reward_repeat_penalty_std": 0.19002924859523773, "reward_total_composite_mean": 0.8288908004760742, "reward_total_composite_std": 0.26403236389160156} {"timestamp_utc": "2026-04-12T01:26:53Z", "mode": "train", "global_step": 2104, "epoch": 0.08450817367554324, "loss": -0.0122, "grad_norm": 1.8657711744308472, "learning_rate": 3.6272727272727275e-06, "num_tokens": 4735591.0, "completions/mean_length": 296.5, "completions/min_length": 284.0, "completions/max_length": 315.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 296.5, "completions/min_terminated_length": 284.0, "completions/max_terminated_length": 315.0, "rewards/meter/mean": 0.9669773578643799, "rewards/meter/std": 0.07153221219778061, "rewards/count_adherence/mean": 0.984375, "rewards/count_adherence/std": 0.04419417306780815, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7205882668495178, "rewards/repeat_penalty/std": 0.061355408281087875, "rewards/total_composite/mean": 0.6847283840179443, "rewards/total_composite/std": 0.07457956671714783, "reward": 0.6847283840179443, "reward_std": 0.07457955926656723, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021201852709054947, "sampling/sampling_logp_difference/max": 1.1567082405090332, "sampling/importance_sampling_ratio/min": 0.3145197927951813, "sampling/importance_sampling_ratio/mean": 1.0043342113494873, "sampling/importance_sampling_ratio/max": 1.9632225036621094, "entropy": 0.17900856165215373, "clip_ratio/low_mean": 0.004951660754159093, "clip_ratio/low_min": 0.004951660754159093, "clip_ratio/high_mean": 0.009039964294061065, "clip_ratio/high_max": 0.009039964294061065, "clip_ratio/region_mean": 0.013991625048220158, "reward_total_mean": 0.6847283840179443, "reward_meter_mean": 0.9669773578643799, "reward_meter_std": 0.07153221219778061, "reward_count_adherence_mean": 0.984375, "reward_count_adherence_std": 0.04419417306780815, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7205882668495178, "reward_repeat_penalty_std": 0.061355408281087875, "reward_total_composite_mean": 0.6847283840179443, "reward_total_composite_std": 0.07457956671714783} {"timestamp_utc": "2026-04-12T01:26:58Z", "mode": "train", "global_step": 2105, "epoch": 0.08454833915732819, "loss": -0.0002, "grad_norm": 4.257723331451416, "learning_rate": 3.6242424242424248e-06, "num_tokens": 4738461.0, "completions/mean_length": 164.75, "completions/min_length": 161.0, "completions/max_length": 166.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 164.75, "completions/min_terminated_length": 161.0, "completions/max_terminated_length": 166.0, "rewards/meter/mean": 0.9989630579948425, "rewards/meter/std": 0.00037122031790204346, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7777777910232544, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7769712209701538, "rewards/total_composite/std": 0.00028871477115899324, "reward": 0.7769712209701538, "reward_std": 0.0002887141308747232, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016738269478082657, "sampling/sampling_logp_difference/max": 2.3132505416870117, "sampling/importance_sampling_ratio/min": 0.09893912822008133, "sampling/importance_sampling_ratio/mean": 1.0012632608413696, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0402666125446558, "clip_ratio/low_mean": 0.004629980307072401, "clip_ratio/low_min": 0.004629980307072401, "clip_ratio/high_mean": 0.009068090235814452, "clip_ratio/high_max": 0.009068090235814452, "clip_ratio/region_mean": 0.013698070542886853, "reward_total_mean": 0.7769712209701538, "reward_meter_mean": 0.9989630579948425, "reward_meter_std": 0.00037122031790204346, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7777777910232544, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7769712209701538, "reward_total_composite_std": 0.00028871477115899324} {"timestamp_utc": "2026-04-12T01:27:06Z", "mode": "train", "global_step": 2106, "epoch": 0.08458850463911315, "loss": -0.0003, "grad_norm": 1.7383067607879639, "learning_rate": 3.6212121212121216e-06, "num_tokens": 4742393.0, "completions/mean_length": 279.5, "completions/min_length": 276.0, "completions/max_length": 282.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 279.5, "completions/min_terminated_length": 276.0, "completions/max_terminated_length": 282.0, "rewards/meter/mean": 0.9945741891860962, "rewards/meter/std": 0.0010695032542571425, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7578125, "rewards/repeat_penalty/std": 0.07790146768093109, "rewards/total_composite/mean": 0.7536386847496033, "rewards/total_composite/std": 0.07672014087438583, "reward": 0.7536386847496033, "reward_std": 0.07672013342380524, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010709014721214771, "sampling/sampling_logp_difference/max": 1.3207725286483765, "sampling/importance_sampling_ratio/min": 0.2669290006160736, "sampling/importance_sampling_ratio/mean": 0.999245285987854, "sampling/importance_sampling_ratio/max": 1.8650121688842773, "entropy": 0.045266281347721815, "clip_ratio/low_mean": 0.00402287530596368, "clip_ratio/low_min": 0.00402287530596368, "clip_ratio/high_mean": 0.00443893379997462, "clip_ratio/high_max": 0.00443893379997462, "clip_ratio/region_mean": 0.0084618091059383, "reward_total_mean": 0.7536386847496033, "reward_meter_mean": 0.9945741891860962, "reward_meter_std": 0.0010695032542571425, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7578125, "reward_repeat_penalty_std": 0.07790146768093109, "reward_total_composite_mean": 0.7536386847496033, "reward_total_composite_std": 0.07672014087438583} {"timestamp_utc": "2026-04-12T01:27:12Z", "mode": "train", "global_step": 2107, "epoch": 0.0846286701208981, "loss": -0.0, "grad_norm": 0.0878043994307518, "learning_rate": 3.6181818181818184e-06, "num_tokens": 4744793.0, "completions/mean_length": 118.0, "completions/min_length": 118.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.0, "completions/min_terminated_length": 118.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.9968082904815674, "rewards/meter/std": 4.3226536945439875e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8544071912765503, "rewards/total_composite/std": 3.7037829315522686e-05, "reward": 0.8544071912765503, "reward_std": 3.7032186810392886e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005184296518564224, "sampling/sampling_logp_difference/max": 1.0821123123168945, "sampling/importance_sampling_ratio/min": 0.33887895941734314, "sampling/importance_sampling_ratio/mean": 0.9996362924575806, "sampling/importance_sampling_ratio/max": 1.1646240949630737, "entropy": 0.026277633383870125, "clip_ratio/low_mean": 0.0021186440717428923, "clip_ratio/low_min": 0.0021186440717428923, "clip_ratio/high_mean": 0.0010593220358714461, "clip_ratio/high_max": 0.0010593220358714461, "clip_ratio/region_mean": 0.0031779661076143384, "reward_total_mean": 0.8544071912765503, "reward_meter_mean": 0.9968082904815674, "reward_meter_std": 4.3226536945439875e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8544071912765503, "reward_total_composite_std": 3.7037829315522686e-05} {"timestamp_utc": "2026-04-12T01:27:17Z", "mode": "train", "global_step": 2108, "epoch": 0.08466883560268305, "loss": -0.0069, "grad_norm": 1.9401229619979858, "learning_rate": 3.6151515151515153e-06, "num_tokens": 4747165.0, "completions/mean_length": 120.5, "completions/min_length": 120.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.5, "completions/min_terminated_length": 120.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.9796810746192932, "rewards/meter/std": 0.003423919202759862, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8397266268730164, "rewards/total_composite/std": 0.002934773452579975, "reward": 0.8397266268730164, "reward_std": 0.00293477950617671, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004073810297995806, "sampling/sampling_logp_difference/max": 0.2728769779205322, "sampling/importance_sampling_ratio/min": 0.761186420917511, "sampling/importance_sampling_ratio/mean": 1.0006637573242188, "sampling/importance_sampling_ratio/max": 1.215042233467102, "entropy": 0.029683739179745317, "clip_ratio/low_mean": 0.004166666883975267, "clip_ratio/low_min": 0.004166666883975267, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.004166666883975267, "reward_total_mean": 0.8397266268730164, "reward_meter_mean": 0.9796810746192932, "reward_meter_std": 0.003423919202759862, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8397266268730164, "reward_total_composite_std": 0.002934773452579975} {"timestamp_utc": "2026-04-12T01:27:22Z", "mode": "train", "global_step": 2109, "epoch": 0.08470900108446801, "loss": 0.0148, "grad_norm": 14.469276428222656, "learning_rate": 3.6121212121212125e-06, "num_tokens": 4748903.0, "completions/mean_length": 59.25, "completions/min_length": 57.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.25, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.675230860710144, "rewards/meter/std": 0.3926205635070801, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.675230860710144, "rewards/total_composite/std": 0.3926205635070801, "reward": 0.675230860710144, "reward_std": 0.3926205635070801, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.032343991100788116, "sampling/sampling_logp_difference/max": 1.3949341773986816, "sampling/importance_sampling_ratio/min": 0.35194310545921326, "sampling/importance_sampling_ratio/mean": 1.008945107460022, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.24129995983093977, "clip_ratio/low_mean": 0.021334989927709103, "clip_ratio/low_min": 0.021334989927709103, "clip_ratio/high_mean": 0.006250000325962901, "clip_ratio/high_max": 0.006250000325962901, "clip_ratio/region_mean": 0.027584990253672004, "reward_total_mean": 0.675230860710144, "reward_meter_mean": 0.675230860710144, "reward_meter_std": 0.3926205635070801, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.675230860710144, "reward_total_composite_std": 0.3926205635070801} {"timestamp_utc": "2026-04-12T01:27:26Z", "mode": "train", "global_step": 2110, "epoch": 0.08474916656625296, "loss": -0.0044, "grad_norm": 0.8487328290939331, "learning_rate": 3.6090909090909093e-06, "num_tokens": 4750550.0, "completions/mean_length": 55.875, "completions/min_length": 55.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.875, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.9948891401290894, "rewards/meter/std": 0.0002349100832361728, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9948891401290894, "rewards/total_composite/std": 0.0002349100832361728, "reward": 0.9948891401290894, "reward_std": 0.00023490135208703578, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005522139370441437, "sampling/sampling_logp_difference/max": 0.4381282329559326, "sampling/importance_sampling_ratio/min": 0.7602133750915527, "sampling/importance_sampling_ratio/mean": 1.0012332201004028, "sampling/importance_sampling_ratio/max": 1.5498037338256836, "entropy": 0.031467124819755554, "clip_ratio/low_mean": 0.0022727272007614374, "clip_ratio/low_min": 0.0022727272007614374, "clip_ratio/high_mean": 0.0022321429569274187, "clip_ratio/high_max": 0.0022321429569274187, "clip_ratio/region_mean": 0.004504870157688856, "reward_total_mean": 0.9948891401290894, "reward_meter_mean": 0.9948891401290894, "reward_meter_std": 0.0002349100832361728, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9948891401290894, "reward_total_composite_std": 0.0002349100832361728} {"timestamp_utc": "2026-04-12T01:27:31Z", "mode": "train", "global_step": 2111, "epoch": 0.08478933204803792, "loss": 0.0071, "grad_norm": 4.360041618347168, "learning_rate": 3.606060606060606e-06, "num_tokens": 4752319.0, "completions/mean_length": 72.125, "completions/min_length": 71.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.125, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9952445030212402, "rewards/meter/std": 0.0019698443356901407, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9952445030212402, "rewards/total_composite/std": 0.0019698443356901407, "reward": 0.9952445030212402, "reward_std": 0.00196986086666584, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.039312709122896194, "sampling/sampling_logp_difference/max": 1.4191043376922607, "sampling/importance_sampling_ratio/min": 0.24193060398101807, "sampling/importance_sampling_ratio/mean": 1.0026034116744995, "sampling/importance_sampling_ratio/max": 1.8677667379379272, "entropy": 0.2009260579943657, "clip_ratio/low_mean": 0.012152777868323028, "clip_ratio/low_min": 0.012152777868323028, "clip_ratio/high_mean": 0.022355403285473585, "clip_ratio/high_max": 0.022355403285473585, "clip_ratio/region_mean": 0.03450818115379661, "reward_total_mean": 0.9952445030212402, "reward_meter_mean": 0.9952445030212402, "reward_meter_std": 0.0019698443356901407, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9952445030212402, "reward_total_composite_std": 0.0019698443356901407} {"timestamp_utc": "2026-04-12T01:27:36Z", "mode": "train", "global_step": 2112, "epoch": 0.08482949752982287, "loss": 0.0141, "grad_norm": 3.6561806201934814, "learning_rate": 3.603030303030303e-06, "num_tokens": 4754067.0, "completions/mean_length": 75.5, "completions/min_length": 74.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.5, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9953248500823975, "rewards/meter/std": 0.002693862421438098, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9953248500823975, "rewards/total_composite/std": 0.002693862421438098, "reward": 0.9953248500823975, "reward_std": 0.002693843562155962, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0394962802529335, "sampling/sampling_logp_difference/max": 1.5090522766113281, "sampling/importance_sampling_ratio/min": 0.2211194485425949, "sampling/importance_sampling_ratio/mean": 1.0075361728668213, "sampling/importance_sampling_ratio/max": 1.8056482076644897, "entropy": 0.28017595037817955, "clip_ratio/low_mean": 0.009980987291783094, "clip_ratio/low_min": 0.009980987291783094, "clip_ratio/high_mean": 0.013099490897729993, "clip_ratio/high_max": 0.013099490897729993, "clip_ratio/region_mean": 0.023080478189513087, "reward_total_mean": 0.9953248500823975, "reward_meter_mean": 0.9953248500823975, "reward_meter_std": 0.002693862421438098, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9953248500823975, "reward_total_composite_std": 0.002693862421438098} {"timestamp_utc": "2026-04-12T01:27:45Z", "mode": "train", "global_step": 2113, "epoch": 0.08486966301160782, "loss": 0.0042, "grad_norm": 1.0902267694473267, "learning_rate": 3.6000000000000003e-06, "num_tokens": 4758194.0, "completions/mean_length": 283.875, "completions/min_length": 254.0, "completions/max_length": 289.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 283.875, "completions/min_terminated_length": 254.0, "completions/max_terminated_length": 289.0, "rewards/meter/mean": 0.9955472946166992, "rewards/meter/std": 0.0001822304038796574, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6993464231491089, "rewards/repeat_penalty/std": 0.05026431009173393, "rewards/total_composite/mean": 0.6962326169013977, "rewards/total_composite/std": 0.05004315823316574, "reward": 0.6962326169013977, "reward_std": 0.05004315450787544, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0065383305773139, "sampling/sampling_logp_difference/max": 0.8383898735046387, "sampling/importance_sampling_ratio/min": 0.4324061870574951, "sampling/importance_sampling_ratio/mean": 0.9997674822807312, "sampling/importance_sampling_ratio/max": 1.2696807384490967, "entropy": 0.03367242659442127, "clip_ratio/low_mean": 0.0008680555620230734, "clip_ratio/low_min": 0.0008680555620230734, "clip_ratio/high_mean": 0.0022863353369757533, "clip_ratio/high_max": 0.0022863353369757533, "clip_ratio/region_mean": 0.0031543908989988267, "reward_total_mean": 0.6962326169013977, "reward_meter_mean": 0.9955472946166992, "reward_meter_std": 0.0001822304038796574, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6993464231491089, "reward_repeat_penalty_std": 0.05026431009173393, "reward_total_composite_mean": 0.6962326169013977, "reward_total_composite_std": 0.05004315823316574} {"timestamp_utc": "2026-04-12T01:27:51Z", "mode": "train", "global_step": 2114, "epoch": 0.08490982849339278, "loss": 0.005, "grad_norm": 2.4666974544525146, "learning_rate": 3.596969696969697e-06, "num_tokens": 4761069.0, "completions/mean_length": 184.375, "completions/min_length": 176.0, "completions/max_length": 193.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 184.375, "completions/min_terminated_length": 176.0, "completions/max_terminated_length": 193.0, "rewards/meter/mean": 0.9969959855079651, "rewards/meter/std": 0.0019897068850696087, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8611111640930176, "rewards/repeat_penalty/std": 0.07856741547584534, "rewards/total_composite/mean": 0.8585879802703857, "rewards/total_composite/std": 0.0791376605629921, "reward": 0.8585879802703857, "reward_std": 0.0791376456618309, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0264764241874218, "sampling/sampling_logp_difference/max": 0.995415210723877, "sampling/importance_sampling_ratio/min": 0.3695699870586395, "sampling/importance_sampling_ratio/mean": 1.0035723447799683, "sampling/importance_sampling_ratio/max": 1.9149926900863647, "entropy": 0.16935125924646854, "clip_ratio/low_mean": 0.003332580265123397, "clip_ratio/low_min": 0.003332580265123397, "clip_ratio/high_mean": 0.01766940689412877, "clip_ratio/high_max": 0.01766940689412877, "clip_ratio/region_mean": 0.021001987159252167, "reward_total_mean": 0.8585879802703857, "reward_meter_mean": 0.9969959855079651, "reward_meter_std": 0.0019897068850696087, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8611111640930176, "reward_repeat_penalty_std": 0.07856741547584534, "reward_total_composite_mean": 0.8585879802703857, "reward_total_composite_std": 0.0791376605629921} {"timestamp_utc": "2026-04-12T01:28:01Z", "mode": "train", "global_step": 2115, "epoch": 0.08494999397517773, "loss": -0.0201, "grad_norm": 1.3121997117996216, "learning_rate": 3.593939393939394e-06, "num_tokens": 4765867.0, "completions/mean_length": 371.75, "completions/min_length": 346.0, "completions/max_length": 399.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 371.75, "completions/min_terminated_length": 346.0, "completions/max_terminated_length": 399.0, "rewards/meter/mean": 0.9931106567382812, "rewards/meter/std": 0.003387609263882041, "rewards/count_adherence/mean": 0.7142857313156128, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7990131378173828, "rewards/repeat_penalty/std": 0.08971592038869858, "rewards/total_composite/mean": 0.5668459534645081, "rewards/total_composite/std": 0.06429950892925262, "reward": 0.5668459534645081, "reward_std": 0.06429950892925262, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03195931762456894, "sampling/sampling_logp_difference/max": 2.8946592807769775, "sampling/importance_sampling_ratio/min": 0.055317867547273636, "sampling/importance_sampling_ratio/mean": 1.0030711889266968, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23509259428828955, "clip_ratio/low_mean": 0.008480485761538148, "clip_ratio/low_min": 0.008480485761538148, "clip_ratio/high_mean": 0.013179366709664464, "clip_ratio/high_max": 0.013179366709664464, "clip_ratio/region_mean": 0.021659852471202612, "reward_total_mean": 0.5668459534645081, "reward_meter_mean": 0.9931106567382812, "reward_meter_std": 0.003387609263882041, "reward_count_adherence_mean": 0.7142857313156128, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7990131378173828, "reward_repeat_penalty_std": 0.08971592038869858, "reward_total_composite_mean": 0.5668459534645081, "reward_total_composite_std": 0.06429950892925262} {"timestamp_utc": "2026-04-12T01:28:07Z", "mode": "train", "global_step": 2116, "epoch": 0.08499015945696269, "loss": 0.0004, "grad_norm": 1.9204028844833374, "learning_rate": 3.590909090909091e-06, "num_tokens": 4768740.0, "completions/mean_length": 165.125, "completions/min_length": 165.0, "completions/max_length": 166.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 165.125, "completions/min_terminated_length": 165.0, "completions/max_terminated_length": 166.0, "rewards/meter/mean": 0.9991816878318787, "rewards/meter/std": 9.469691576668993e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.819444477558136, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.8187738060951233, "rewards/total_composite/std": 0.057456936687231064, "reward": 0.8187738060951233, "reward_std": 0.05745694413781166, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009189879521727562, "sampling/sampling_logp_difference/max": 1.0871639251708984, "sampling/importance_sampling_ratio/min": 0.33717140555381775, "sampling/importance_sampling_ratio/mean": 0.9997456669807434, "sampling/importance_sampling_ratio/max": 1.796520471572876, "entropy": 0.04427940887399018, "clip_ratio/low_mean": 0.0015151514671742916, "clip_ratio/low_min": 0.0015151514671742916, "clip_ratio/high_mean": 0.0015060240402817726, "clip_ratio/high_max": 0.0015060240402817726, "clip_ratio/region_mean": 0.0030211755074560642, "reward_total_mean": 0.8187738060951233, "reward_meter_mean": 0.9991816878318787, "reward_meter_std": 9.469691576668993e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.819444477558136, "reward_repeat_penalty_std": 0.05750546231865883, "reward_total_composite_mean": 0.8187738060951233, "reward_total_composite_std": 0.057456936687231064} {"timestamp_utc": "2026-04-12T01:28:14Z", "mode": "train", "global_step": 2117, "epoch": 0.08503032493874764, "loss": 0.0088, "grad_norm": 1.891755223274231, "learning_rate": 3.587878787878788e-06, "num_tokens": 4771637.0, "completions/mean_length": 182.125, "completions/min_length": 178.0, "completions/max_length": 186.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 182.125, "completions/min_terminated_length": 178.0, "completions/max_terminated_length": 186.0, "rewards/meter/mean": 0.9905871152877808, "rewards/meter/std": 0.00902599561959505, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8055555820465088, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.7980640530586243, "rewards/total_composite/std": 0.05313723534345627, "reward": 0.7980640530586243, "reward_std": 0.053137242794036865, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018866930156946182, "sampling/sampling_logp_difference/max": 1.2814984321594238, "sampling/importance_sampling_ratio/min": 0.27762100100517273, "sampling/importance_sampling_ratio/mean": 1.0019904375076294, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09395178686827421, "clip_ratio/low_mean": 0.010289846803061664, "clip_ratio/low_min": 0.010289846803061664, "clip_ratio/high_mean": 0.006211843807250261, "clip_ratio/high_max": 0.006211843807250261, "clip_ratio/region_mean": 0.016501690610311925, "reward_total_mean": 0.7980640530586243, "reward_meter_mean": 0.9905871152877808, "reward_meter_std": 0.00902599561959505, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8055555820465088, "reward_repeat_penalty_std": 0.05143444985151291, "reward_total_composite_mean": 0.7980640530586243, "reward_total_composite_std": 0.05313723534345627} {"timestamp_utc": "2026-04-12T01:28:22Z", "mode": "train", "global_step": 2118, "epoch": 0.0850704904205326, "loss": -0.0196, "grad_norm": 1.509689211845398, "learning_rate": 3.584848484848485e-06, "num_tokens": 4775851.0, "completions/mean_length": 297.75, "completions/min_length": 279.0, "completions/max_length": 312.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 297.75, "completions/min_terminated_length": 279.0, "completions/max_terminated_length": 312.0, "rewards/meter/mean": 0.99297034740448, "rewards/meter/std": 0.005611373111605644, "rewards/count_adherence/mean": 0.8888888955116272, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8921875357627869, "rewards/repeat_penalty/std": 0.1552397757768631, "rewards/total_composite/mean": 0.7875304222106934, "rewards/total_composite/std": 0.13774561882019043, "reward": 0.7875304222106934, "reward_std": 0.13774563372135162, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04715384542942047, "sampling/sampling_logp_difference/max": 1.5841054916381836, "sampling/importance_sampling_ratio/min": 0.20513120293617249, "sampling/importance_sampling_ratio/mean": 1.00875985622406, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.424762025475502, "clip_ratio/low_mean": 0.00718731596134603, "clip_ratio/low_min": 0.00718731596134603, "clip_ratio/high_mean": 0.02530877012759447, "clip_ratio/high_max": 0.02530877012759447, "clip_ratio/region_mean": 0.0324960860889405, "reward_total_mean": 0.7875304222106934, "reward_meter_mean": 0.99297034740448, "reward_meter_std": 0.005611373111605644, "reward_count_adherence_mean": 0.8888888955116272, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8921875357627869, "reward_repeat_penalty_std": 0.1552397757768631, "reward_total_composite_mean": 0.7875304222106934, "reward_total_composite_std": 0.13774561882019043} {"timestamp_utc": "2026-04-12T01:28:27Z", "mode": "train", "global_step": 2119, "epoch": 0.08511065590231755, "loss": -0.004, "grad_norm": 3.0825119018554688, "learning_rate": 3.5818181818181817e-06, "num_tokens": 4777544.0, "completions/mean_length": 55.625, "completions/min_length": 54.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.625, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.9949027895927429, "rewards/meter/std": 0.00035745359491556883, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9949027895927429, "rewards/total_composite/std": 0.00035745359491556883, "reward": 0.9949027895927429, "reward_std": 0.00035746910725720227, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01183076947927475, "sampling/sampling_logp_difference/max": 0.6624326705932617, "sampling/importance_sampling_ratio/min": 0.515595555305481, "sampling/importance_sampling_ratio/mean": 0.9991430640220642, "sampling/importance_sampling_ratio/max": 1.3233060836791992, "entropy": 0.08348313788883388, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.004545454401522875, "clip_ratio/high_max": 0.004545454401522875, "clip_ratio/region_mean": 0.004545454401522875, "reward_total_mean": 0.9949027895927429, "reward_meter_mean": 0.9949027895927429, "reward_meter_std": 0.00035745359491556883, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9949027895927429, "reward_total_composite_std": 0.00035745359491556883} {"timestamp_utc": "2026-04-12T01:28:32Z", "mode": "train", "global_step": 2120, "epoch": 0.0851508213841025, "loss": 0.0047, "grad_norm": 3.3534421920776367, "learning_rate": 3.578787878787879e-06, "num_tokens": 4779365.0, "completions/mean_length": 73.625, "completions/min_length": 72.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.625, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9964693784713745, "rewards/meter/std": 0.002390751615166664, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9964693784713745, "rewards/total_composite/std": 0.002390751615166664, "reward": 0.9964693784713745, "reward_std": 0.002390736248344183, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03976722061634064, "sampling/sampling_logp_difference/max": 1.3943734169006348, "sampling/importance_sampling_ratio/min": 0.24798838794231415, "sampling/importance_sampling_ratio/mean": 0.9964186549186707, "sampling/importance_sampling_ratio/max": 1.3884204626083374, "entropy": 0.3053772822022438, "clip_ratio/low_mean": 0.015343417064286768, "clip_ratio/low_min": 0.015343417064286768, "clip_ratio/high_mean": 0.020273972768336535, "clip_ratio/high_max": 0.020273972768336535, "clip_ratio/region_mean": 0.0356173898326233, "reward_total_mean": 0.9964693784713745, "reward_meter_mean": 0.9964693784713745, "reward_meter_std": 0.002390751615166664, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9964693784713745, "reward_total_composite_std": 0.002390751615166664} {"timestamp_utc": "2026-04-12T01:28:36Z", "mode": "train", "global_step": 2121, "epoch": 0.08519098686588746, "loss": -0.0199, "grad_norm": 4.196082592010498, "learning_rate": 3.575757575757576e-06, "num_tokens": 4781079.0, "completions/mean_length": 55.25, "completions/min_length": 52.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.25, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.994369626045227, "rewards/meter/std": 0.0011783665977418423, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.994369626045227, "rewards/total_composite/std": 0.0011783665977418423, "reward": 0.994369626045227, "reward_std": 0.0011783792870119214, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00968268234282732, "sampling/sampling_logp_difference/max": 0.9451119899749756, "sampling/importance_sampling_ratio/min": 0.7468781471252441, "sampling/importance_sampling_ratio/mean": 1.0018255710601807, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.053725593723356724, "clip_ratio/low_mean": 0.002314814832061529, "clip_ratio/low_min": 0.002314814832061529, "clip_ratio/high_mean": 0.0022321429569274187, "clip_ratio/high_max": 0.0022321429569274187, "clip_ratio/region_mean": 0.004546957788988948, "reward_total_mean": 0.994369626045227, "reward_meter_mean": 0.994369626045227, "reward_meter_std": 0.0011783665977418423, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.994369626045227, "reward_total_composite_std": 0.0011783665977418423} {"timestamp_utc": "2026-04-12T01:28:45Z", "mode": "train", "global_step": 2122, "epoch": 0.08523115234767241, "loss": -0.0042, "grad_norm": 1.6565197706222534, "learning_rate": 3.5727272727272734e-06, "num_tokens": 4786520.0, "completions/mean_length": 410.125, "completions/min_length": 383.0, "completions/max_length": 429.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 410.125, "completions/min_terminated_length": 383.0, "completions/max_terminated_length": 429.0, "rewards/meter/mean": 0.9933267831802368, "rewards/meter/std": 0.007835081778466702, "rewards/count_adherence/mean": 0.671875, "rewards/count_adherence/std": 0.0289318785071373, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.82259202003479, "rewards/repeat_penalty/std": 0.032492659986019135, "rewards/total_composite/mean": 0.5489572286605835, "rewards/total_composite/std": 0.031959354877471924, "reward": 0.5489572286605835, "reward_std": 0.031959328800439835, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030476173385977745, "sampling/sampling_logp_difference/max": 4.228629112243652, "sampling/importance_sampling_ratio/min": 0.014572354033589363, "sampling/importance_sampling_ratio/mean": 1.0041075944900513, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.21682057529687881, "clip_ratio/low_mean": 0.009774382109753788, "clip_ratio/low_min": 0.009774382109753788, "clip_ratio/high_mean": 0.01417637791018933, "clip_ratio/high_max": 0.01417637791018933, "clip_ratio/region_mean": 0.023950760019943118, "reward_total_mean": 0.5489572286605835, "reward_meter_mean": 0.9933267831802368, "reward_meter_std": 0.007835081778466702, "reward_count_adherence_mean": 0.671875, "reward_count_adherence_std": 0.0289318785071373, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.82259202003479, "reward_repeat_penalty_std": 0.032492659986019135, "reward_total_composite_mean": 0.5489572286605835, "reward_total_composite_std": 0.031959354877471924} {"timestamp_utc": "2026-04-12T01:28:50Z", "mode": "train", "global_step": 2123, "epoch": 0.08527131782945736, "loss": 0.0001, "grad_norm": 0.5491177439689636, "learning_rate": 3.5696969696969703e-06, "num_tokens": 4788264.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.999085009098053, "rewards/meter/std": 1.2789209904440213e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999085009098053, "rewards/total_composite/std": 1.2789209904440213e-05, "reward": 0.999085009098053, "reward_std": 1.2776405128533952e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004210304468870163, "sampling/sampling_logp_difference/max": 0.254666805267334, "sampling/importance_sampling_ratio/min": 0.8502018451690674, "sampling/importance_sampling_ratio/mean": 1.0031498670578003, "sampling/importance_sampling_ratio/max": 1.2900317907333374, "entropy": 0.04304229188710451, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0018939394503831863, "clip_ratio/high_max": 0.0018939394503831863, "clip_ratio/region_mean": 0.0018939394503831863, "reward_total_mean": 0.999085009098053, "reward_meter_mean": 0.999085009098053, "reward_meter_std": 1.2789209904440213e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.999085009098053, "reward_total_composite_std": 1.2789209904440213e-05} {"timestamp_utc": "2026-04-12T01:28:55Z", "mode": "train", "global_step": 2124, "epoch": 0.08531148331124232, "loss": 0.0436, "grad_norm": 5.764961242675781, "learning_rate": 3.566666666666667e-06, "num_tokens": 4790353.0, "completions/mean_length": 82.125, "completions/min_length": 80.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.125, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.4380970001220703, "rewards/meter/std": 0.3100660443305969, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4380970001220703, "rewards/total_composite/std": 0.3100660443305969, "reward": 0.4380970001220703, "reward_std": 0.3100660443305969, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030523981899023056, "sampling/sampling_logp_difference/max": 1.2927680015563965, "sampling/importance_sampling_ratio/min": 0.27450987696647644, "sampling/importance_sampling_ratio/mean": 1.0140416622161865, "sampling/importance_sampling_ratio/max": 1.603717565536499, "entropy": 0.26322684064507484, "clip_ratio/low_mean": 0.017462147399783134, "clip_ratio/low_min": 0.017462147399783134, "clip_ratio/high_mean": 0.0031250000465661287, "clip_ratio/high_max": 0.0031250000465661287, "clip_ratio/region_mean": 0.020587147446349263, "reward_total_mean": 0.4380970001220703, "reward_meter_mean": 0.4380970001220703, "reward_meter_std": 0.3100660443305969, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.4380970001220703, "reward_total_composite_std": 0.3100660443305969} {"timestamp_utc": "2026-04-12T01:29:00Z", "mode": "train", "global_step": 2125, "epoch": 0.08535164879302727, "loss": 0.0064, "grad_norm": 1.9213858842849731, "learning_rate": 3.563636363636364e-06, "num_tokens": 4793170.0, "completions/mean_length": 156.125, "completions/min_length": 154.0, "completions/max_length": 159.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 156.125, "completions/min_terminated_length": 154.0, "completions/max_terminated_length": 159.0, "rewards/meter/mean": 0.9884169101715088, "rewards/meter/std": 0.011561714112758636, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7777777910232544, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7687686681747437, "rewards/total_composite/std": 0.008992438204586506, "reward": 0.7687686681747437, "reward_std": 0.008992421440780163, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008227693848311901, "sampling/sampling_logp_difference/max": 1.741776704788208, "sampling/importance_sampling_ratio/min": 0.17520882189273834, "sampling/importance_sampling_ratio/mean": 1.000543475151062, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03148621576838195, "clip_ratio/low_mean": 0.0007861634949222207, "clip_ratio/low_min": 0.0007861634949222207, "clip_ratio/high_mean": 0.0072432131273671985, "clip_ratio/high_max": 0.0072432131273671985, "clip_ratio/region_mean": 0.00802937662228942, "reward_total_mean": 0.7687686681747437, "reward_meter_mean": 0.9884169101715088, "reward_meter_std": 0.011561714112758636, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7777777910232544, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7687686681747437, "reward_total_composite_std": 0.008992438204586506} {"timestamp_utc": "2026-04-12T01:29:05Z", "mode": "train", "global_step": 2126, "epoch": 0.08539181427481222, "loss": -0.0063, "grad_norm": 6.276111125946045, "learning_rate": 3.560606060606061e-06, "num_tokens": 4794531.0, "completions/mean_length": 33.125, "completions/min_length": 33.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.125, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9689929485321045, "rewards/meter/std": 0.0023667889181524515, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9689929485321045, "rewards/total_composite/std": 0.0023667889181524515, "reward": 0.9689929485321045, "reward_std": 0.0023667817004024982, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0230213962495327, "sampling/sampling_logp_difference/max": 0.7616815567016602, "sampling/importance_sampling_ratio/min": 0.4668806791305542, "sampling/importance_sampling_ratio/mean": 1.0059295892715454, "sampling/importance_sampling_ratio/max": 1.4770609140396118, "entropy": 0.14932595752179623, "clip_ratio/low_mean": 0.018939394503831863, "clip_ratio/low_min": 0.018939394503831863, "clip_ratio/high_mean": 0.018605169840157032, "clip_ratio/high_max": 0.018605169840157032, "clip_ratio/region_mean": 0.037544564343988895, "reward_total_mean": 0.9689929485321045, "reward_meter_mean": 0.9689929485321045, "reward_meter_std": 0.0023667889181524515, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9689929485321045, "reward_total_composite_std": 0.0023667889181524515} {"timestamp_utc": "2026-04-12T01:29:09Z", "mode": "train", "global_step": 2127, "epoch": 0.08543197975659718, "loss": 0.0053, "grad_norm": 3.2434964179992676, "learning_rate": 3.557575757575758e-06, "num_tokens": 4796260.0, "completions/mean_length": 54.125, "completions/min_length": 54.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.125, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.7054018974304199, "rewards/meter/std": 0.08442744612693787, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7054018974304199, "rewards/total_composite/std": 0.08442744612693787, "reward": 0.7054018974304199, "reward_std": 0.08442744612693787, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009495479986071587, "sampling/sampling_logp_difference/max": 0.5380792617797852, "sampling/importance_sampling_ratio/min": 0.5838686227798462, "sampling/importance_sampling_ratio/mean": 1.0031954050064087, "sampling/importance_sampling_ratio/max": 1.4377515316009521, "entropy": 0.049557043705135584, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0069444444961845875, "clip_ratio/high_max": 0.0069444444961845875, "clip_ratio/region_mean": 0.0069444444961845875, "reward_total_mean": 0.7054018974304199, "reward_meter_mean": 0.7054018974304199, "reward_meter_std": 0.08442744612693787, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7054018974304199, "reward_total_composite_std": 0.08442744612693787} {"timestamp_utc": "2026-04-12T01:29:16Z", "mode": "train", "global_step": 2128, "epoch": 0.08547214523838213, "loss": -0.0058, "grad_norm": 2.1206531524658203, "learning_rate": 3.554545454545455e-06, "num_tokens": 4799957.0, "completions/mean_length": 261.125, "completions/min_length": 248.0, "completions/max_length": 268.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 261.125, "completions/min_terminated_length": 248.0, "completions/max_terminated_length": 268.0, "rewards/meter/mean": 0.9929981231689453, "rewards/meter/std": 0.0028132039587944746, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8942307829856873, "rewards/repeat_penalty/std": 0.05723259970545769, "rewards/total_composite/mean": 0.8880020976066589, "rewards/total_composite/std": 0.05752360075712204, "reward": 0.8880020976066589, "reward_std": 0.05752360448241234, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04205208271741867, "sampling/sampling_logp_difference/max": 0.9664134979248047, "sampling/importance_sampling_ratio/min": 0.38044506311416626, "sampling/importance_sampling_ratio/mean": 1.0112462043762207, "sampling/importance_sampling_ratio/max": 1.9891395568847656, "entropy": 0.43808240443468094, "clip_ratio/low_mean": 0.009215709753334522, "clip_ratio/low_min": 0.009215709753334522, "clip_ratio/high_mean": 0.02282354235649109, "clip_ratio/high_max": 0.02282354235649109, "clip_ratio/region_mean": 0.03203925210982561, "reward_total_mean": 0.8880020976066589, "reward_meter_mean": 0.9929981231689453, "reward_meter_std": 0.0028132039587944746, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8942307829856873, "reward_repeat_penalty_std": 0.05723259970545769, "reward_total_composite_mean": 0.8880020976066589, "reward_total_composite_std": 0.05752360075712204} {"timestamp_utc": "2026-04-12T01:29:23Z", "mode": "train", "global_step": 2129, "epoch": 0.08551231072016709, "loss": -0.0205, "grad_norm": 2.319450855255127, "learning_rate": 3.551515151515152e-06, "num_tokens": 4803610.0, "completions/mean_length": 264.625, "completions/min_length": 252.0, "completions/max_length": 281.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 264.625, "completions/min_terminated_length": 252.0, "completions/max_terminated_length": 281.0, "rewards/meter/mean": 0.9950700998306274, "rewards/meter/std": 0.0019416395807638764, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8949176073074341, "rewards/repeat_penalty/std": 0.04042290523648262, "rewards/total_composite/mean": 0.8904905915260315, "rewards/total_composite/std": 0.039885781705379486, "reward": 0.8904905915260315, "reward_std": 0.039885781705379486, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.040030911564826965, "sampling/sampling_logp_difference/max": 3.957237720489502, "sampling/importance_sampling_ratio/min": 0.01911584474146366, "sampling/importance_sampling_ratio/mean": 1.0098717212677002, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.35909293219447136, "clip_ratio/low_mean": 0.01029538200236857, "clip_ratio/low_min": 0.01029538200236857, "clip_ratio/high_mean": 0.02008880244102329, "clip_ratio/high_max": 0.02008880244102329, "clip_ratio/region_mean": 0.03038418444339186, "reward_total_mean": 0.8904905915260315, "reward_meter_mean": 0.9950700998306274, "reward_meter_std": 0.0019416395807638764, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8949176073074341, "reward_repeat_penalty_std": 0.04042290523648262, "reward_total_composite_mean": 0.8904905915260315, "reward_total_composite_std": 0.039885781705379486} {"timestamp_utc": "2026-04-12T01:29:28Z", "mode": "train", "global_step": 2130, "epoch": 0.08555247620195204, "loss": 0.0131, "grad_norm": 4.817035675048828, "learning_rate": 3.548484848484849e-06, "num_tokens": 4805553.0, "completions/mean_length": 70.875, "completions/min_length": 68.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.875, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9955647587776184, "rewards/meter/std": 0.0008866624557413161, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9955647587776184, "rewards/total_composite/std": 0.0008866624557413161, "reward": 0.9955647587776184, "reward_std": 0.0008866624557413161, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.037295129150152206, "sampling/sampling_logp_difference/max": 1.1011011600494385, "sampling/importance_sampling_ratio/min": 0.3325047492980957, "sampling/importance_sampling_ratio/mean": 1.006104826927185, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.20211379043757915, "clip_ratio/low_mean": 0.017461657291278243, "clip_ratio/low_min": 0.017461657291278243, "clip_ratio/high_mean": 0.01770029927138239, "clip_ratio/high_max": 0.01770029927138239, "clip_ratio/region_mean": 0.035161956562660635, "reward_total_mean": 0.9955647587776184, "reward_meter_mean": 0.9955647587776184, "reward_meter_std": 0.0008866624557413161, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9955647587776184, "reward_total_composite_std": 0.0008866624557413161} {"timestamp_utc": "2026-04-12T01:29:34Z", "mode": "train", "global_step": 2131, "epoch": 0.085592641683737, "loss": -0.0121, "grad_norm": 2.6760542392730713, "learning_rate": 3.5454545454545458e-06, "num_tokens": 4808185.0, "completions/mean_length": 142.0, "completions/min_length": 138.0, "completions/max_length": 146.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 142.0, "completions/min_terminated_length": 138.0, "completions/max_terminated_length": 146.0, "rewards/meter/mean": 0.9969066381454468, "rewards/meter/std": 0.0009874114766716957, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9969066381454468, "rewards/total_composite/std": 0.0009874114766716957, "reward": 0.9969066381454468, "reward_std": 0.0009874175302684307, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03481653332710266, "sampling/sampling_logp_difference/max": 1.2930517196655273, "sampling/importance_sampling_ratio/min": 0.2744320034980774, "sampling/importance_sampling_ratio/mean": 1.0037314891815186, "sampling/importance_sampling_ratio/max": 1.913059115409851, "entropy": 0.28764577955007553, "clip_ratio/low_mean": 0.015936652896925807, "clip_ratio/low_min": 0.015936652896925807, "clip_ratio/high_mean": 0.014961764682084322, "clip_ratio/high_max": 0.014961764682084322, "clip_ratio/region_mean": 0.03089841757901013, "reward_total_mean": 0.9969066381454468, "reward_meter_mean": 0.9969066381454468, "reward_meter_std": 0.0009874114766716957, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9969066381454468, "reward_total_composite_std": 0.0009874114766716957} {"timestamp_utc": "2026-04-12T01:29:39Z", "mode": "train", "global_step": 2132, "epoch": 0.08563280716552195, "loss": 0.0002, "grad_norm": 3.024273633956909, "learning_rate": 3.5424242424242426e-06, "num_tokens": 4810360.0, "completions/mean_length": 107.875, "completions/min_length": 106.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.875, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.9939501285552979, "rewards/meter/std": 0.003658889327198267, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9690351486206055, "rewards/total_composite/std": 0.06948735564947128, "reward": 0.9690351486206055, "reward_std": 0.06948734074831009, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03451121971011162, "sampling/sampling_logp_difference/max": 1.3759374618530273, "sampling/importance_sampling_ratio/min": 0.2526026964187622, "sampling/importance_sampling_ratio/mean": 1.0037057399749756, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17346674110740423, "clip_ratio/low_mean": 0.004672897048294544, "clip_ratio/low_min": 0.004672897048294544, "clip_ratio/high_mean": 0.026603603502735496, "clip_ratio/high_max": 0.026603603502735496, "clip_ratio/region_mean": 0.03127650055103004, "reward_total_mean": 0.9690351486206055, "reward_meter_mean": 0.9939501285552979, "reward_meter_std": 0.003658889327198267, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9690351486206055, "reward_total_composite_std": 0.06948735564947128} {"timestamp_utc": "2026-04-12T01:29:45Z", "mode": "train", "global_step": 2133, "epoch": 0.0856729726473069, "loss": 0.001, "grad_norm": 0.6826779246330261, "learning_rate": 3.53939393939394e-06, "num_tokens": 4813068.0, "completions/mean_length": 165.5, "completions/min_length": 165.0, "completions/max_length": 167.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 165.5, "completions/min_terminated_length": 165.0, "completions/max_terminated_length": 167.0, "rewards/meter/mean": 0.999198317527771, "rewards/meter/std": 3.6679448385257274e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7916666865348816, "rewards/repeat_penalty/std": 0.03928370773792267, "rewards/total_composite/mean": 0.7910317182540894, "rewards/total_composite/std": 0.039245039224624634, "reward": 0.7910317182540894, "reward_std": 0.03924502432346344, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008064919151365757, "sampling/sampling_logp_difference/max": 1.644291639328003, "sampling/importance_sampling_ratio/min": 0.19314932823181152, "sampling/importance_sampling_ratio/mean": 1.0013805627822876, "sampling/importance_sampling_ratio/max": 1.5342899560928345, "entropy": 0.03646064782515168, "clip_ratio/low_mean": 0.0030303029343485832, "clip_ratio/low_min": 0.0030303029343485832, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0030303029343485832, "reward_total_mean": 0.7910317182540894, "reward_meter_mean": 0.999198317527771, "reward_meter_std": 3.6679448385257274e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7916666865348816, "reward_repeat_penalty_std": 0.03928370773792267, "reward_total_composite_mean": 0.7910317182540894, "reward_total_composite_std": 0.039245039224624634} {"timestamp_utc": "2026-04-12T01:29:49Z", "mode": "train", "global_step": 2134, "epoch": 0.08571313812909186, "loss": 0.0003, "grad_norm": 0.44613105058670044, "learning_rate": 3.5363636363636367e-06, "num_tokens": 4814740.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.7352948188781738, "rewards/meter/std": 3.051780367968604e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7352948188781738, "rewards/total_composite/std": 3.051780367968604e-05, "reward": 0.7352948188781738, "reward_std": 3.0517016057274304e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005636686459183693, "sampling/sampling_logp_difference/max": 0.6817889213562012, "sampling/importance_sampling_ratio/min": 0.7626789808273315, "sampling/importance_sampling_ratio/mean": 1.0043625831604004, "sampling/importance_sampling_ratio/max": 1.9774119853973389, "entropy": 0.037100087851285934, "clip_ratio/low_mean": 0.004629629664123058, "clip_ratio/low_min": 0.004629629664123058, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.004629629664123058, "reward_total_mean": 0.7352948188781738, "reward_meter_mean": 0.7352948188781738, "reward_meter_std": 3.051780367968604e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7352948188781738, "reward_total_composite_std": 3.051780367968604e-05} {"timestamp_utc": "2026-04-12T01:29:56Z", "mode": "train", "global_step": 2135, "epoch": 0.08575330361087681, "loss": 0.0222, "grad_norm": 4.601060390472412, "learning_rate": 3.5333333333333335e-06, "num_tokens": 4818135.0, "completions/mean_length": 229.375, "completions/min_length": 224.0, "completions/max_length": 240.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 229.375, "completions/min_terminated_length": 224.0, "completions/max_terminated_length": 240.0, "rewards/meter/mean": 0.971871018409729, "rewards/meter/std": 0.06841260194778442, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9431818723678589, "rewards/repeat_penalty/std": 0.047049909830093384, "rewards/total_composite/mean": 0.9174596071243286, "rewards/total_composite/std": 0.08795089274644852, "reward": 0.9174596071243286, "reward_std": 0.08795087039470673, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.041487276554107666, "sampling/sampling_logp_difference/max": 1.7106046676635742, "sampling/importance_sampling_ratio/min": 0.18075646460056305, "sampling/importance_sampling_ratio/mean": 1.0070955753326416, "sampling/importance_sampling_ratio/max": 1.8775804042816162, "entropy": 0.4052262455224991, "clip_ratio/low_mean": 0.016251615015789866, "clip_ratio/low_min": 0.016251615015789866, "clip_ratio/high_mean": 0.014304422307759523, "clip_ratio/high_max": 0.014304422307759523, "clip_ratio/region_mean": 0.03055603732354939, "reward_total_mean": 0.9174596071243286, "reward_meter_mean": 0.971871018409729, "reward_meter_std": 0.06841260194778442, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9431818723678589, "reward_repeat_penalty_std": 0.047049909830093384, "reward_total_composite_mean": 0.9174596071243286, "reward_total_composite_std": 0.08795089274644852} {"timestamp_utc": "2026-04-12T01:30:02Z", "mode": "train", "global_step": 2136, "epoch": 0.08579346909266176, "loss": -0.0019, "grad_norm": 1.9712090492248535, "learning_rate": 3.5303030303030304e-06, "num_tokens": 4820878.0, "completions/mean_length": 165.875, "completions/min_length": 165.0, "completions/max_length": 167.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 165.875, "completions/min_terminated_length": 165.0, "completions/max_terminated_length": 167.0, "rewards/meter/mean": 0.9991864562034607, "rewards/meter/std": 2.615616722323466e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8611111640930176, "rewards/repeat_penalty/std": 0.07856741547584534, "rewards/total_composite/mean": 0.8604106903076172, "rewards/total_composite/std": 0.07850649952888489, "reward": 0.8604106903076172, "reward_std": 0.07850649952888489, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009452839381992817, "sampling/sampling_logp_difference/max": 0.7467837333679199, "sampling/importance_sampling_ratio/min": 0.5129733085632324, "sampling/importance_sampling_ratio/mean": 1.0031864643096924, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06440420588478446, "clip_ratio/low_mean": 0.004527199547737837, "clip_ratio/low_min": 0.004527199547737837, "clip_ratio/high_mean": 0.0015015150420367718, "clip_ratio/high_max": 0.0015015150420367718, "clip_ratio/region_mean": 0.006028714589774609, "reward_total_mean": 0.8604106903076172, "reward_meter_mean": 0.9991864562034607, "reward_meter_std": 2.615616722323466e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8611111640930176, "reward_repeat_penalty_std": 0.07856741547584534, "reward_total_composite_mean": 0.8604106903076172, "reward_total_composite_std": 0.07850649952888489} {"timestamp_utc": "2026-04-12T01:30:08Z", "mode": "train", "global_step": 2137, "epoch": 0.08583363457444672, "loss": -0.0007, "grad_norm": 1.4629026651382446, "learning_rate": 3.5272727272727276e-06, "num_tokens": 4823547.0, "completions/mean_length": 145.625, "completions/min_length": 145.0, "completions/max_length": 148.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 145.625, "completions/min_terminated_length": 145.0, "completions/max_terminated_length": 148.0, "rewards/meter/mean": 0.9696428775787354, "rewards/meter/std": 0.0028882811311632395, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7777777910232544, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7541667222976685, "rewards/total_composite/std": 0.0022464508656412363, "reward": 0.7541667222976685, "reward_std": 0.0022464566864073277, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011477367021143436, "sampling/sampling_logp_difference/max": 1.31986665725708, "sampling/importance_sampling_ratio/min": 0.26717090606689453, "sampling/importance_sampling_ratio/mean": 1.0017269849777222, "sampling/importance_sampling_ratio/max": 1.7878315448760986, "entropy": 0.04937975388020277, "clip_ratio/low_mean": 0.006010864395648241, "clip_ratio/low_min": 0.006010864395648241, "clip_ratio/high_mean": 0.0025862068869173527, "clip_ratio/high_max": 0.0025862068869173527, "clip_ratio/region_mean": 0.008597071282565594, "reward_total_mean": 0.7541667222976685, "reward_meter_mean": 0.9696428775787354, "reward_meter_std": 0.0028882811311632395, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7777777910232544, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7541667222976685, "reward_total_composite_std": 0.0022464508656412363} {"timestamp_utc": "2026-04-12T01:30:15Z", "mode": "train", "global_step": 2138, "epoch": 0.08587380005623167, "loss": 0.0015, "grad_norm": 2.5688741207122803, "learning_rate": 3.5242424242424244e-06, "num_tokens": 4826897.0, "completions/mean_length": 228.75, "completions/min_length": 225.0, "completions/max_length": 233.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 228.75, "completions/min_terminated_length": 225.0, "completions/max_terminated_length": 233.0, "rewards/meter/mean": 0.9738757610321045, "rewards/meter/std": 0.055281247943639755, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9659091234207153, "rewards/repeat_penalty/std": 0.047049909830093384, "rewards/total_composite/mean": 0.9418025016784668, "rewards/total_composite/std": 0.08390164375305176, "reward": 0.9418025016784668, "reward_std": 0.08390163630247116, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04527745395898819, "sampling/sampling_logp_difference/max": 1.5867786407470703, "sampling/importance_sampling_ratio/min": 0.20458358526229858, "sampling/importance_sampling_ratio/mean": 1.0137181282043457, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40121684968471527, "clip_ratio/low_mean": 0.01095851231366396, "clip_ratio/low_min": 0.01095851231366396, "clip_ratio/high_mean": 0.02402331749908626, "clip_ratio/high_max": 0.02402331749908626, "clip_ratio/region_mean": 0.03498182981275022, "reward_total_mean": 0.9418025016784668, "reward_meter_mean": 0.9738757610321045, "reward_meter_std": 0.055281247943639755, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9659091234207153, "reward_repeat_penalty_std": 0.047049909830093384, "reward_total_composite_mean": 0.9418025016784668, "reward_total_composite_std": 0.08390164375305176} {"timestamp_utc": "2026-04-12T01:30:20Z", "mode": "train", "global_step": 2139, "epoch": 0.08591396553801663, "loss": 0.0018, "grad_norm": 1.6966145038604736, "learning_rate": 3.5212121212121213e-06, "num_tokens": 4828577.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.7483930587768555, "rewards/meter/std": 0.02422301471233368, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7483930587768555, "rewards/total_composite/std": 0.02422301471233368, "reward": 0.7483930587768555, "reward_std": 0.024223024025559425, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005149809177964926, "sampling/sampling_logp_difference/max": 0.4711747169494629, "sampling/importance_sampling_ratio/min": 0.6358863711357117, "sampling/importance_sampling_ratio/mean": 1.0017740726470947, "sampling/importance_sampling_ratio/max": 1.6018749475479126, "entropy": 0.020970646291971207, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.002314814832061529, "clip_ratio/high_max": 0.002314814832061529, "clip_ratio/region_mean": 0.002314814832061529, "reward_total_mean": 0.7483930587768555, "reward_meter_mean": 0.7483930587768555, "reward_meter_std": 0.02422301471233368, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7483930587768555, "reward_total_composite_std": 0.02422301471233368} {"timestamp_utc": "2026-04-12T01:30:25Z", "mode": "train", "global_step": 2140, "epoch": 0.08595413101980158, "loss": -0.0033, "grad_norm": 3.241760015487671, "learning_rate": 3.5181818181818185e-06, "num_tokens": 4830436.0, "completions/mean_length": 76.375, "completions/min_length": 73.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.375, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9935569763183594, "rewards/meter/std": 0.004046422429382801, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9935569763183594, "rewards/total_composite/std": 0.004046422429382801, "reward": 0.9935569763183594, "reward_std": 0.004046411253511906, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03363337367773056, "sampling/sampling_logp_difference/max": 1.207101821899414, "sampling/importance_sampling_ratio/min": 0.2990627586841583, "sampling/importance_sampling_ratio/mean": 1.008273720741272, "sampling/importance_sampling_ratio/max": 1.7292462587356567, "entropy": 0.3370419256389141, "clip_ratio/low_mean": 0.004959081998094916, "clip_ratio/low_min": 0.004959081998094916, "clip_ratio/high_mean": 0.02094770153053105, "clip_ratio/high_max": 0.02094770153053105, "clip_ratio/region_mean": 0.025906783528625965, "reward_total_mean": 0.9935569763183594, "reward_meter_mean": 0.9935569763183594, "reward_meter_std": 0.004046422429382801, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9935569763183594, "reward_total_composite_std": 0.004046422429382801} {"timestamp_utc": "2026-04-12T01:30:29Z", "mode": "train", "global_step": 2141, "epoch": 0.08599429650158653, "loss": 0.0014, "grad_norm": 4.896651268005371, "learning_rate": 3.5151515151515154e-06, "num_tokens": 4832051.0, "completions/mean_length": 55.875, "completions/min_length": 55.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.875, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.9941670894622803, "rewards/meter/std": 0.002000106731429696, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9941670894622803, "rewards/total_composite/std": 0.002000106731429696, "reward": 0.9941670894622803, "reward_std": 0.0020000922959297895, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011897793039679527, "sampling/sampling_logp_difference/max": 0.8048620223999023, "sampling/importance_sampling_ratio/min": 0.4471496343612671, "sampling/importance_sampling_ratio/mean": 1.0028151273727417, "sampling/importance_sampling_ratio/max": 1.2935843467712402, "entropy": 0.0797986383549869, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.004464285913854837, "clip_ratio/high_max": 0.004464285913854837, "clip_ratio/region_mean": 0.004464285913854837, "reward_total_mean": 0.9941670894622803, "reward_meter_mean": 0.9941670894622803, "reward_meter_std": 0.002000106731429696, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9941670894622803, "reward_total_composite_std": 0.002000106731429696} {"timestamp_utc": "2026-04-12T01:30:39Z", "mode": "train", "global_step": 2142, "epoch": 0.08603446198337149, "loss": 0.013, "grad_norm": 1.6217327117919922, "learning_rate": 3.512121212121212e-06, "num_tokens": 4837339.0, "completions/mean_length": 428.0, "completions/min_length": 411.0, "completions/max_length": 448.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 428.0, "completions/min_terminated_length": 411.0, "completions/max_terminated_length": 448.0, "rewards/meter/mean": 0.9931625127792358, "rewards/meter/std": 0.0038301812019199133, "rewards/count_adherence/mean": 0.6953125, "rewards/count_adherence/std": 0.022097086533904076, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8884575366973877, "rewards/repeat_penalty/std": 0.061322376132011414, "rewards/total_composite/mean": 0.6134706139564514, "rewards/total_composite/std": 0.04515310749411583, "reward": 0.6134706139564514, "reward_std": 0.04515310376882553, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.042900450527668, "sampling/sampling_logp_difference/max": 2.926726818084717, "sampling/importance_sampling_ratio/min": 0.05357210338115692, "sampling/importance_sampling_ratio/mean": 1.0085182189941406, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3651280999183655, "clip_ratio/low_mean": 0.005864104256033897, "clip_ratio/low_min": 0.005864104256033897, "clip_ratio/high_mean": 0.027495570946484804, "clip_ratio/high_max": 0.027495570946484804, "clip_ratio/region_mean": 0.0333596752025187, "reward_total_mean": 0.6134706139564514, "reward_meter_mean": 0.9931625127792358, "reward_meter_std": 0.0038301812019199133, "reward_count_adherence_mean": 0.6953125, "reward_count_adherence_std": 0.022097086533904076, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8884575366973877, "reward_repeat_penalty_std": 0.061322376132011414, "reward_total_composite_mean": 0.6134706139564514, "reward_total_composite_std": 0.04515310749411583} {"timestamp_utc": "2026-04-12T01:30:43Z", "mode": "train", "global_step": 2143, "epoch": 0.08607462746515644, "loss": 0.0064, "grad_norm": 5.659821510314941, "learning_rate": 3.509090909090909e-06, "num_tokens": 4839153.0, "completions/mean_length": 74.75, "completions/min_length": 72.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.75, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9972976446151733, "rewards/meter/std": 0.0019415906863287091, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9972976446151733, "rewards/total_composite/std": 0.0019415906863287091, "reward": 0.9972976446151733, "reward_std": 0.0019415918504819274, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06912153214216232, "sampling/sampling_logp_difference/max": 1.5144741535186768, "sampling/importance_sampling_ratio/min": 0.4317888617515564, "sampling/importance_sampling_ratio/mean": 1.0229203701019287, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6021371968090534, "clip_ratio/low_mean": 0.040365297347307205, "clip_ratio/low_min": 0.040365297347307205, "clip_ratio/high_mean": 0.024497864302247763, "clip_ratio/high_max": 0.024497864302247763, "clip_ratio/region_mean": 0.06486316164955497, "reward_total_mean": 0.9972976446151733, "reward_meter_mean": 0.9972976446151733, "reward_meter_std": 0.0019415906863287091, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9972976446151733, "reward_total_composite_std": 0.0019415906863287091} {"timestamp_utc": "2026-04-12T01:30:48Z", "mode": "train", "global_step": 2144, "epoch": 0.0861147929469414, "loss": -0.0009, "grad_norm": 5.293501853942871, "learning_rate": 3.5060606060606063e-06, "num_tokens": 4841063.0, "completions/mean_length": 74.75, "completions/min_length": 72.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.75, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9942959547042847, "rewards/meter/std": 0.0059119779616594315, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9942959547042847, "rewards/total_composite/std": 0.0059119779616594315, "reward": 0.9942959547042847, "reward_std": 0.005911969114094973, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06987594068050385, "sampling/sampling_logp_difference/max": 2.395887851715088, "sampling/importance_sampling_ratio/min": 0.09109176695346832, "sampling/importance_sampling_ratio/mean": 1.0201735496520996, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.48777300119400024, "clip_ratio/low_mean": 0.021914413664489985, "clip_ratio/low_min": 0.021914413664489985, "clip_ratio/high_mean": 0.025277130538597703, "clip_ratio/high_max": 0.025277130538597703, "clip_ratio/region_mean": 0.04719154420308769, "reward_total_mean": 0.9942959547042847, "reward_meter_mean": 0.9942959547042847, "reward_meter_std": 0.0059119779616594315, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9942959547042847, "reward_total_composite_std": 0.0059119779616594315} {"timestamp_utc": "2026-04-12T01:30:52Z", "mode": "train", "global_step": 2145, "epoch": 0.08615495842872635, "loss": 0.0132, "grad_norm": 7.162275314331055, "learning_rate": 3.503030303030303e-06, "num_tokens": 4842409.0, "completions/mean_length": 29.25, "completions/min_length": 29.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.25, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.9901993274688721, "rewards/meter/std": 0.004359308164566755, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9901993274688721, "rewards/total_composite/std": 0.004359308164566755, "reward": 0.9901993274688721, "reward_std": 0.0043593174777925014, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01579299196600914, "sampling/sampling_logp_difference/max": 0.4930543899536133, "sampling/importance_sampling_ratio/min": 0.6107580661773682, "sampling/importance_sampling_ratio/mean": 0.9988601207733154, "sampling/importance_sampling_ratio/max": 1.2712383270263672, "entropy": 0.09351515863090754, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.01278735650703311, "clip_ratio/high_max": 0.01278735650703311, "clip_ratio/region_mean": 0.01278735650703311, "reward_total_mean": 0.9901993274688721, "reward_meter_mean": 0.9901993274688721, "reward_meter_std": 0.004359308164566755, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9901993274688721, "reward_total_composite_std": 0.004359308164566755} {"timestamp_utc": "2026-04-12T01:31:01Z", "mode": "train", "global_step": 2146, "epoch": 0.0861951239105113, "loss": 0.5018, "grad_norm": 5.25761604309082, "learning_rate": 3.5e-06, "num_tokens": 4845311.0, "completions/mean_length": 198.75, "completions/min_length": 143.0, "completions/max_length": 488.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 198.75, "completions/min_terminated_length": 143.0, "completions/max_terminated_length": 488.0, "rewards/meter/mean": 0.8164126873016357, "rewards/meter/std": 0.32973745465278625, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.26726123690605164, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.9434523582458496, "rewards/repeat_penalty/std": 0.0783882662653923, "rewards/total_composite/mean": 0.7369348406791687, "rewards/total_composite/std": 0.32373327016830444, "reward": 0.7369348406791687, "reward_std": 0.32373327016830444, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09025763720273972, "sampling/sampling_logp_difference/max": 1.6007208824157715, "sampling/importance_sampling_ratio/min": 0.20175102353096008, "sampling/importance_sampling_ratio/mean": 1.013696312904358, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2716873697936535, "clip_ratio/low_mean": 0.008386766072362661, "clip_ratio/low_min": 0.008386766072362661, "clip_ratio/high_mean": 0.02845356403850019, "clip_ratio/high_max": 0.02845356403850019, "clip_ratio/region_mean": 0.03684033011086285, "reward_total_mean": 0.7369348406791687, "reward_meter_mean": 0.8164126873016357, "reward_meter_std": 0.32973745465278625, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.26726123690605164, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.9434523582458496, "reward_repeat_penalty_std": 0.0783882662653923, "reward_total_composite_mean": 0.7369348406791687, "reward_total_composite_std": 0.32373327016830444} {"timestamp_utc": "2026-04-12T01:31:06Z", "mode": "train", "global_step": 2147, "epoch": 0.08623528939229626, "loss": -0.0056, "grad_norm": 7.887787342071533, "learning_rate": 3.496969696969697e-06, "num_tokens": 4847185.0, "completions/mean_length": 73.25, "completions/min_length": 69.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.25, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.8275381922721863, "rewards/meter/std": 0.3271496891975403, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8275381922721863, "rewards/total_composite/std": 0.3271496891975403, "reward": 0.8275381922721863, "reward_std": 0.3271496891975403, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0584087148308754, "sampling/sampling_logp_difference/max": 1.7163329124450684, "sampling/importance_sampling_ratio/min": 0.17972400784492493, "sampling/importance_sampling_ratio/mean": 1.0108306407928467, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.41511600464582443, "clip_ratio/low_mean": 0.00345633109100163, "clip_ratio/low_min": 0.00345633109100163, "clip_ratio/high_mean": 0.029611698118969798, "clip_ratio/high_max": 0.029611698118969798, "clip_ratio/region_mean": 0.03306802920997143, "reward_total_mean": 0.8275381922721863, "reward_meter_mean": 0.8275381922721863, "reward_meter_std": 0.3271496891975403, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8275381922721863, "reward_total_composite_std": 0.3271496891975403} {"timestamp_utc": "2026-04-12T01:31:10Z", "mode": "train", "global_step": 2148, "epoch": 0.08627545487408121, "loss": -0.0013, "grad_norm": 6.055057525634766, "learning_rate": 3.493939393939394e-06, "num_tokens": 4848601.0, "completions/mean_length": 29.0, "completions/min_length": 29.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9926831722259521, "rewards/meter/std": 0.0014186145272105932, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9926831722259521, "rewards/total_composite/std": 0.0014186145272105932, "reward": 0.9926831722259521, "reward_std": 0.0014186184853315353, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018664227798581123, "sampling/sampling_logp_difference/max": 0.8764760494232178, "sampling/importance_sampling_ratio/min": 0.4162471890449524, "sampling/importance_sampling_ratio/mean": 0.9984536170959473, "sampling/importance_sampling_ratio/max": 1.2208077907562256, "entropy": 0.0592546658590436, "clip_ratio/low_mean": 0.004310344811528921, "clip_ratio/low_min": 0.004310344811528921, "clip_ratio/high_mean": 0.012931034434586763, "clip_ratio/high_max": 0.012931034434586763, "clip_ratio/region_mean": 0.017241379246115685, "reward_total_mean": 0.9926831722259521, "reward_meter_mean": 0.9926831722259521, "reward_meter_std": 0.0014186145272105932, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9926831722259521, "reward_total_composite_std": 0.0014186145272105932} {"timestamp_utc": "2026-04-12T01:31:16Z", "mode": "train", "global_step": 2149, "epoch": 0.08631562035586617, "loss": 0.012, "grad_norm": 3.5529558658599854, "learning_rate": 3.4909090909090913e-06, "num_tokens": 4851588.0, "completions/mean_length": 186.375, "completions/min_length": 181.0, "completions/max_length": 196.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 186.375, "completions/min_terminated_length": 181.0, "completions/max_terminated_length": 196.0, "rewards/meter/mean": 0.9706813097000122, "rewards/meter/std": 0.06584999710321426, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9166666269302368, "rewards/repeat_penalty/std": 0.07856741547584534, "rewards/total_composite/mean": 0.8901457786560059, "rewards/total_composite/std": 0.10105938464403152, "reward": 0.8901457786560059, "reward_std": 0.10105939954519272, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05906498804688454, "sampling/sampling_logp_difference/max": 1.6412153244018555, "sampling/importance_sampling_ratio/min": 0.19374443590641022, "sampling/importance_sampling_ratio/mean": 1.0136346817016602, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5572481229901314, "clip_ratio/low_mean": 0.021443208097480237, "clip_ratio/low_min": 0.021443208097480237, "clip_ratio/high_mean": 0.02710459241643548, "clip_ratio/high_max": 0.02710459241643548, "clip_ratio/region_mean": 0.04854780051391572, "reward_total_mean": 0.8901457786560059, "reward_meter_mean": 0.9706813097000122, "reward_meter_std": 0.06584999710321426, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9166666269302368, "reward_repeat_penalty_std": 0.07856741547584534, "reward_total_composite_mean": 0.8901457786560059, "reward_total_composite_std": 0.10105938464403152} {"timestamp_utc": "2026-04-12T01:31:22Z", "mode": "train", "global_step": 2150, "epoch": 0.08635578583765112, "loss": 0.0075, "grad_norm": 2.0675745010375977, "learning_rate": 3.4878787878787885e-06, "num_tokens": 4854574.0, "completions/mean_length": 179.25, "completions/min_length": 174.0, "completions/max_length": 182.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 179.25, "completions/min_terminated_length": 174.0, "completions/max_terminated_length": 182.0, "rewards/meter/mean": 0.9972676634788513, "rewards/meter/std": 0.0014413135359063745, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9444444179534912, "rewards/repeat_penalty/std": 0.08399210125207901, "rewards/total_composite/mean": 0.9418361783027649, "rewards/total_composite/std": 0.08344186097383499, "reward": 0.9418361783027649, "reward_std": 0.0834418535232544, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02511134371161461, "sampling/sampling_logp_difference/max": 1.1651017665863037, "sampling/importance_sampling_ratio/min": 0.3118909001350403, "sampling/importance_sampling_ratio/mean": 1.0084565877914429, "sampling/importance_sampling_ratio/max": 1.8844407796859741, "entropy": 0.2794443480670452, "clip_ratio/low_mean": 0.006343001441564411, "clip_ratio/low_min": 0.006343001441564411, "clip_ratio/high_mean": 0.009757570689544082, "clip_ratio/high_max": 0.009757570689544082, "clip_ratio/region_mean": 0.016100572131108493, "reward_total_mean": 0.9418361783027649, "reward_meter_mean": 0.9972676634788513, "reward_meter_std": 0.0014413135359063745, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9444444179534912, "reward_repeat_penalty_std": 0.08399210125207901, "reward_total_composite_mean": 0.9418361783027649, "reward_total_composite_std": 0.08344186097383499} {"timestamp_utc": "2026-04-12T01:32:33Z", "mode": "eval", "global_step": 2150, "epoch": 0.08635578583765112, "eval_loss": NaN, "eval_runtime": 71.0335, "eval_samples_per_second": 1.464, "eval_steps_per_second": 0.183, "eval_num_tokens": 4854574.0, "eval_completions/mean_length": 208.15384615384616, "eval_completions/min_length": 61.23076923076923, "eval_completions/max_length": 377.61538461538464, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 208.15384615384616, "eval_completions/min_terminated_length": 61.23076923076923, "eval_completions/max_terminated_length": 377.61538461538464, "eval_rewards/meter/mean": 0.7046629740641668, "eval_rewards/meter/std": 0.41940173277488124, "eval_rewards/count_adherence/mean": 0.9253275165191064, "eval_rewards/count_adherence/std": 0.10161341182314433, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/repeat_penalty/mean": 0.843710142832536, "eval_rewards/repeat_penalty/std": 0.15254983076682457, "eval_rewards/total_composite/mean": 0.5750110378632178, "eval_rewards/total_composite/std": 0.38267656473013073, "eval_reward": 0.5750110378632178, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.02115864851153814, "eval_sampling/sampling_logp_difference/max": 1.116743463736314, "eval_sampling/importance_sampling_ratio/min": 0.3407066028851729, "eval_sampling/importance_sampling_ratio/mean": 1.0052896371254554, "eval_sampling/importance_sampling_ratio/max": 1.5623481090252216, "eval_entropy": 0.20996456879835862, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5750110378632178, "eval_reward_meter_mean": 0.7046629740641668, "eval_reward_meter_std": 0.41940173277488124, "eval_reward_count_adherence_mean": 0.9253275165191064, "eval_reward_count_adherence_std": 0.10161341182314433, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_repeat_penalty_mean": 0.843710142832536, "eval_reward_repeat_penalty_std": 0.15254983076682457, "eval_reward_total_composite_mean": 0.5750110378632178, "eval_reward_total_composite_std": 0.38267656473013073} {"timestamp_utc": "2026-04-12T01:32:46Z", "mode": "train", "global_step": 2151, "epoch": 0.08639595131943607, "loss": -0.0636, "grad_norm": 3.3067500591278076, "learning_rate": 3.4848484848484854e-06, "num_tokens": 4856155.0, "completions/mean_length": 101.625, "completions/min_length": 41.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 43.000003814697266, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.4682069420814514, "rewards/meter/std": 0.35492372512817383, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4681951105594635, "rewards/total_composite/std": 0.35494157671928406, "reward": 0.4681951105594635, "reward_std": 0.35494154691696167, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11403389275074005, "sampling/sampling_logp_difference/max": 2.580704689025879, "sampling/importance_sampling_ratio/min": 0.07572062313556671, "sampling/importance_sampling_ratio/mean": 1.0368061065673828, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9084643498063087, "clip_ratio/low_mean": 0.06512659136205912, "clip_ratio/low_min": 0.06512659136205912, "clip_ratio/high_mean": 0.023466601967811584, "clip_ratio/high_max": 0.023466601967811584, "clip_ratio/region_mean": 0.0885931933298707, "reward_total_mean": 0.4681951105594635, "reward_meter_mean": 0.4682069420814514, "reward_meter_std": 0.35492372512817383, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.4681951105594635, "reward_total_composite_std": 0.35494157671928406} {"timestamp_utc": "2026-04-12T01:32:53Z", "mode": "train", "global_step": 2152, "epoch": 0.08643611680122103, "loss": 0.0533, "grad_norm": 6.426714897155762, "learning_rate": 3.481818181818182e-06, "num_tokens": 4859871.0, "completions/mean_length": 247.5, "completions/min_length": 238.0, "completions/max_length": 276.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 247.5, "completions/min_terminated_length": 238.0, "completions/max_terminated_length": 276.0, "rewards/meter/mean": 0.9982793927192688, "rewards/meter/std": 0.00048778695054352283, "rewards/count_adherence/mean": 0.9642857313156128, "rewards/count_adherence/std": 0.06613000482320786, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8910256624221802, "rewards/repeat_penalty/std": 0.06309091299772263, "rewards/total_composite/mean": 0.8597774505615234, "rewards/total_composite/std": 0.10398055613040924, "reward": 0.8597774505615234, "reward_std": 0.10398054867982864, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028371067717671394, "sampling/sampling_logp_difference/max": 3.265172243118286, "sampling/importance_sampling_ratio/min": 0.03819035738706589, "sampling/importance_sampling_ratio/mean": 1.0031156539916992, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1678062528371811, "clip_ratio/low_mean": 0.010588002507574856, "clip_ratio/low_min": 0.010588002507574856, "clip_ratio/high_mean": 0.0109593840315938, "clip_ratio/high_max": 0.0109593840315938, "clip_ratio/region_mean": 0.021547386539168656, "reward_total_mean": 0.8597774505615234, "reward_meter_mean": 0.9982793927192688, "reward_meter_std": 0.00048778695054352283, "reward_count_adherence_mean": 0.9642857313156128, "reward_count_adherence_std": 0.06613000482320786, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8910256624221802, "reward_repeat_penalty_std": 0.06309091299772263, "reward_total_composite_mean": 0.8597774505615234, "reward_total_composite_std": 0.10398055613040924} {"timestamp_utc": "2026-04-12T01:32:57Z", "mode": "train", "global_step": 2153, "epoch": 0.08647628228300598, "loss": 0.0, "grad_norm": 0.24814273416996002, "learning_rate": 3.4787878787878795e-06, "num_tokens": 4861871.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9990808367729187, "rewards/meter/std": 1.3548809874919243e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990808367729187, "rewards/total_composite/std": 1.3548809874919243e-05, "reward": 0.9990808367729187, "reward_std": 1.3551140909839887e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011720891110599041, "sampling/sampling_logp_difference/max": 1.327371597290039, "sampling/importance_sampling_ratio/min": 0.2651733458042145, "sampling/importance_sampling_ratio/mean": 1.0004948377609253, "sampling/importance_sampling_ratio/max": 1.5080136060714722, "entropy": 0.07956603448837996, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/high_mean": 0.0037878789007663727, "clip_ratio/high_max": 0.0037878789007663727, "clip_ratio/region_mean": 0.007575757801532745, "reward_total_mean": 0.9990808367729187, "reward_meter_mean": 0.9990808367729187, "reward_meter_std": 1.3548809874919243e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990808367729187, "reward_total_composite_std": 1.3548809874919243e-05} {"timestamp_utc": "2026-04-12T01:33:02Z", "mode": "train", "global_step": 2154, "epoch": 0.08651644776479094, "loss": 0.0058, "grad_norm": 3.136981248855591, "learning_rate": 3.4757575757575763e-06, "num_tokens": 4863701.0, "completions/mean_length": 72.75, "completions/min_length": 71.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.75, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9944970607757568, "rewards/meter/std": 0.004257019143551588, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9944970607757568, "rewards/total_composite/std": 0.004257019143551588, "reward": 0.9944970607757568, "reward_std": 0.004257019609212875, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03498362377285957, "sampling/sampling_logp_difference/max": 1.4066009521484375, "sampling/importance_sampling_ratio/min": 0.2449745386838913, "sampling/importance_sampling_ratio/mean": 1.0077894926071167, "sampling/importance_sampling_ratio/max": 1.6701947450637817, "entropy": 0.23252216633409262, "clip_ratio/low_mean": 0.011893743183463812, "clip_ratio/low_min": 0.011893743183463812, "clip_ratio/high_mean": 0.022406263276934624, "clip_ratio/high_max": 0.022406263276934624, "clip_ratio/region_mean": 0.034300006460398436, "reward_total_mean": 0.9944970607757568, "reward_meter_mean": 0.9944970607757568, "reward_meter_std": 0.004257019143551588, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9944970607757568, "reward_total_composite_std": 0.004257019143551588} {"timestamp_utc": "2026-04-12T01:33:06Z", "mode": "train", "global_step": 2155, "epoch": 0.08655661324657589, "loss": -0.0003, "grad_norm": 0.40986379981040955, "learning_rate": 3.472727272727273e-06, "num_tokens": 4865461.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9990643262863159, "rewards/meter/std": 2.489090729795862e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990643262863159, "rewards/total_composite/std": 2.489090729795862e-05, "reward": 0.9990643262863159, "reward_std": 2.4881244826246984e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005097586195915937, "sampling/sampling_logp_difference/max": 0.27621328830718994, "sampling/importance_sampling_ratio/min": 0.8046210408210754, "sampling/importance_sampling_ratio/mean": 1.0042916536331177, "sampling/importance_sampling_ratio/max": 1.318129062652588, "entropy": 0.07073036208748817, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/high_mean": 0.0037878789007663727, "clip_ratio/high_max": 0.0037878789007663727, "clip_ratio/region_mean": 0.005681818351149559, "reward_total_mean": 0.9990643262863159, "reward_meter_mean": 0.9990643262863159, "reward_meter_std": 2.489090729795862e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990643262863159, "reward_total_composite_std": 2.489090729795862e-05} {"timestamp_utc": "2026-04-12T01:33:12Z", "mode": "train", "global_step": 2156, "epoch": 0.08659677872836084, "loss": -0.0024, "grad_norm": 1.6892346143722534, "learning_rate": 3.46969696969697e-06, "num_tokens": 4867682.0, "completions/mean_length": 117.625, "completions/min_length": 116.0, "completions/max_length": 119.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 117.625, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 119.0, "rewards/meter/mean": 0.9963239431381226, "rewards/meter/std": 0.0006225909455679357, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9429587125778198, "rewards/total_composite/std": 0.07379542291164398, "reward": 0.9429587125778198, "reward_std": 0.07379541546106339, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020759226754307747, "sampling/sampling_logp_difference/max": 1.4962215423583984, "sampling/importance_sampling_ratio/min": 0.22397485375404358, "sampling/importance_sampling_ratio/mean": 1.0034620761871338, "sampling/importance_sampling_ratio/max": 1.3534270524978638, "entropy": 0.13127075601369143, "clip_ratio/low_mean": 0.0032144944416359067, "clip_ratio/low_min": 0.0032144944416359067, "clip_ratio/high_mean": 0.009525301051326096, "clip_ratio/high_max": 0.009525301051326096, "clip_ratio/region_mean": 0.012739795492962003, "reward_total_mean": 0.9429587125778198, "reward_meter_mean": 0.9963239431381226, "reward_meter_std": 0.0006225909455679357, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.9429587125778198, "reward_total_composite_std": 0.07379542291164398} {"timestamp_utc": "2026-04-12T01:33:16Z", "mode": "train", "global_step": 2157, "epoch": 0.0866369442101458, "loss": 0.0158, "grad_norm": 1.6599547863006592, "learning_rate": 3.4666666666666672e-06, "num_tokens": 4869399.0, "completions/mean_length": 56.625, "completions/min_length": 54.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.625, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.725243330001831, "rewards/meter/std": 0.03141331672668457, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.725243330001831, "rewards/total_composite/std": 0.03141331672668457, "reward": 0.725243330001831, "reward_std": 0.03141331672668457, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01150722336024046, "sampling/sampling_logp_difference/max": 0.5583809614181519, "sampling/importance_sampling_ratio/min": 0.5721346139907837, "sampling/importance_sampling_ratio/mean": 1.0009208917617798, "sampling/importance_sampling_ratio/max": 1.4941579103469849, "entropy": 0.054770099464803934, "clip_ratio/low_mean": 0.004385964944958687, "clip_ratio/low_min": 0.004385964944958687, "clip_ratio/high_mean": 0.00889376224949956, "clip_ratio/high_max": 0.00889376224949956, "clip_ratio/region_mean": 0.013279727194458246, "reward_total_mean": 0.725243330001831, "reward_meter_mean": 0.725243330001831, "reward_meter_std": 0.03141331672668457, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.725243330001831, "reward_total_composite_std": 0.03141331672668457} {"timestamp_utc": "2026-04-12T01:33:20Z", "mode": "train", "global_step": 2158, "epoch": 0.08667710969193075, "loss": 0.0002, "grad_norm": 0.1878003627061844, "learning_rate": 3.463636363636364e-06, "num_tokens": 4871311.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9990821480751038, "rewards/meter/std": 1.2787881132680923e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990821480751038, "rewards/total_composite/std": 1.2787881132680923e-05, "reward": 0.9990821480751038, "reward_std": 1.2793108908226714e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008319023996591568, "sampling/sampling_logp_difference/max": 1.0997838973999023, "sampling/importance_sampling_ratio/min": 0.33294302225112915, "sampling/importance_sampling_ratio/mean": 1.0013294219970703, "sampling/importance_sampling_ratio/max": 1.3374110460281372, "entropy": 0.07476874813437462, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/high_mean": 0.005681818351149559, "clip_ratio/high_max": 0.005681818351149559, "clip_ratio/region_mean": 0.009469697251915932, "reward_total_mean": 0.9990821480751038, "reward_meter_mean": 0.9990821480751038, "reward_meter_std": 1.2787881132680923e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990821480751038, "reward_total_composite_std": 1.2787881132680923e-05} {"timestamp_utc": "2026-04-12T01:33:25Z", "mode": "train", "global_step": 2159, "epoch": 0.0867172751737157, "loss": 0.0063, "grad_norm": 3.910881280899048, "learning_rate": 3.460606060606061e-06, "num_tokens": 4873104.0, "completions/mean_length": 56.125, "completions/min_length": 56.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.125, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9947191476821899, "rewards/meter/std": 0.0006506324862129986, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9947191476821899, "rewards/total_composite/std": 0.0006506324862129986, "reward": 0.9947191476821899, "reward_std": 0.0006506353965960443, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009572095237672329, "sampling/sampling_logp_difference/max": 0.4652571678161621, "sampling/importance_sampling_ratio/min": 0.6279736161231995, "sampling/importance_sampling_ratio/mean": 1.002916693687439, "sampling/importance_sampling_ratio/max": 1.261582612991333, "entropy": 0.093318160623312, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0022321429569274187, "clip_ratio/high_max": 0.0022321429569274187, "clip_ratio/region_mean": 0.0022321429569274187, "reward_total_mean": 0.9947191476821899, "reward_meter_mean": 0.9947191476821899, "reward_meter_std": 0.0006506324862129986, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9947191476821899, "reward_total_composite_std": 0.0006506324862129986} {"timestamp_utc": "2026-04-12T01:33:29Z", "mode": "train", "global_step": 2160, "epoch": 0.08675744065550066, "loss": 0.0003, "grad_norm": 1.6953576803207397, "learning_rate": 3.4575757575757577e-06, "num_tokens": 4875040.0, "completions/mean_length": 87.0, "completions/min_length": 86.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.0, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.9958928227424622, "rewards/meter/std": 0.000379539153072983, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9958928227424622, "rewards/total_composite/std": 0.000379539153072983, "reward": 0.9958928227424622, "reward_std": 0.000379539153072983, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018249215558171272, "sampling/sampling_logp_difference/max": 0.9508843421936035, "sampling/importance_sampling_ratio/min": 0.38639914989471436, "sampling/importance_sampling_ratio/mean": 1.0018218755722046, "sampling/importance_sampling_ratio/max": 1.3476742506027222, "entropy": 0.11874265689402819, "clip_ratio/low_mean": 0.004310344811528921, "clip_ratio/low_min": 0.004310344811528921, "clip_ratio/high_mean": 0.01438528229482472, "clip_ratio/high_max": 0.01438528229482472, "clip_ratio/region_mean": 0.01869562710635364, "reward_total_mean": 0.9958928227424622, "reward_meter_mean": 0.9958928227424622, "reward_meter_std": 0.000379539153072983, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9958928227424622, "reward_total_composite_std": 0.000379539153072983} {"timestamp_utc": "2026-04-12T01:33:34Z", "mode": "train", "global_step": 2161, "epoch": 0.08679760613728561, "loss": 0.0029, "grad_norm": 2.359445810317993, "learning_rate": 3.454545454545455e-06, "num_tokens": 4876737.0, "completions/mean_length": 56.125, "completions/min_length": 56.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.125, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9948000907897949, "rewards/meter/std": 0.0004447593819350004, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9948000907897949, "rewards/total_composite/std": 0.0004447593819350004, "reward": 0.9948000907897949, "reward_std": 0.00044475641334429383, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007567924913018942, "sampling/sampling_logp_difference/max": 0.8509788513183594, "sampling/importance_sampling_ratio/min": 0.42699676752090454, "sampling/importance_sampling_ratio/mean": 1.0003728866577148, "sampling/importance_sampling_ratio/max": 1.2896134853363037, "entropy": 0.06367568951100111, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.006696428870782256, "clip_ratio/high_max": 0.006696428870782256, "clip_ratio/region_mean": 0.006696428870782256, "reward_total_mean": 0.9948000907897949, "reward_meter_mean": 0.9948000907897949, "reward_meter_std": 0.0004447593819350004, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9948000907897949, "reward_total_composite_std": 0.0004447593819350004} {"timestamp_utc": "2026-04-12T01:33:39Z", "mode": "train", "global_step": 2162, "epoch": 0.08683777161907057, "loss": -0.0273, "grad_norm": 4.7188615798950195, "learning_rate": 3.451515151515152e-06, "num_tokens": 4878413.0, "completions/mean_length": 53.5, "completions/min_length": 49.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.5, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.7199991941452026, "rewards/meter/std": 0.19119207561016083, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.709714949131012, "rewards/total_composite/std": 0.2202804535627365, "reward": 0.709714949131012, "reward_std": 0.22028043866157532, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014019476249814034, "sampling/sampling_logp_difference/max": 1.523421287536621, "sampling/importance_sampling_ratio/min": 0.21796490252017975, "sampling/importance_sampling_ratio/mean": 0.9992796778678894, "sampling/importance_sampling_ratio/max": 1.4815089702606201, "entropy": 0.08292486914433539, "clip_ratio/low_mean": 0.0025510203558951616, "clip_ratio/low_min": 0.0025510203558951616, "clip_ratio/high_mean": 0.0022727272007614374, "clip_ratio/high_max": 0.0022727272007614374, "clip_ratio/region_mean": 0.004823747556656599, "reward_total_mean": 0.709714949131012, "reward_meter_mean": 0.7199991941452026, "reward_meter_std": 0.19119207561016083, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.709714949131012, "reward_total_composite_std": 0.2202804535627365} {"timestamp_utc": "2026-04-12T01:33:45Z", "mode": "train", "global_step": 2163, "epoch": 0.08687793710085552, "loss": 0.0136, "grad_norm": 2.0742194652557373, "learning_rate": 3.4484848484848486e-06, "num_tokens": 4882602.0, "completions/mean_length": 254.625, "completions/min_length": 248.0, "completions/max_length": 272.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 254.625, "completions/min_terminated_length": 248.0, "completions/max_terminated_length": 272.0, "rewards/meter/mean": 0.9978955984115601, "rewards/meter/std": 0.0005918731912970543, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9340659379959106, "rewards/repeat_penalty/std": 0.062163230031728745, "rewards/total_composite/mean": 0.8155835270881653, "rewards/total_composite/std": 0.05420287698507309, "reward": 0.8155835270881653, "reward_std": 0.054202865809202194, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04149610176682472, "sampling/sampling_logp_difference/max": 16.053680419921875, "sampling/importance_sampling_ratio/min": 1.0665352334626732e-07, "sampling/importance_sampling_ratio/mean": 1.0046485662460327, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.32685666531324387, "clip_ratio/low_mean": 0.00980698096100241, "clip_ratio/low_min": 0.00980698096100241, "clip_ratio/high_mean": 0.011834641918540001, "clip_ratio/high_max": 0.011834641918540001, "clip_ratio/region_mean": 0.02164162287954241, "reward_total_mean": 0.8155835270881653, "reward_meter_mean": 0.9978955984115601, "reward_meter_std": 0.0005918731912970543, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9340659379959106, "reward_repeat_penalty_std": 0.062163230031728745, "reward_total_composite_mean": 0.8155835270881653, "reward_total_composite_std": 0.05420287698507309} {"timestamp_utc": "2026-04-12T01:33:50Z", "mode": "train", "global_step": 2164, "epoch": 0.08691810258264047, "loss": -0.007, "grad_norm": 3.962695837020874, "learning_rate": 3.445454545454546e-06, "num_tokens": 4884392.0, "completions/mean_length": 67.75, "completions/min_length": 65.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.75, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9326764345169067, "rewards/meter/std": 0.024432551115751266, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9326764345169067, "rewards/total_composite/std": 0.024432551115751266, "reward": 0.9326764345169067, "reward_std": 0.024432554841041565, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021882515400648117, "sampling/sampling_logp_difference/max": 1.0846776962280273, "sampling/importance_sampling_ratio/min": 0.3380107283592224, "sampling/importance_sampling_ratio/mean": 1.0046216249465942, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.14008523616939783, "clip_ratio/low_mean": 0.00378874852322042, "clip_ratio/low_min": 0.00378874852322042, "clip_ratio/high_mean": 0.0055147059028968215, "clip_ratio/high_max": 0.0055147059028968215, "clip_ratio/region_mean": 0.009303454426117241, "reward_total_mean": 0.9326764345169067, "reward_meter_mean": 0.9326764345169067, "reward_meter_std": 0.024432551115751266, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9326764345169067, "reward_total_composite_std": 0.024432551115751266} {"timestamp_utc": "2026-04-12T01:33:55Z", "mode": "train", "global_step": 2165, "epoch": 0.08695826806442543, "loss": -0.0061, "grad_norm": 3.4584054946899414, "learning_rate": 3.4424242424242427e-06, "num_tokens": 4886154.0, "completions/mean_length": 66.25, "completions/min_length": 66.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.25, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9357738494873047, "rewards/meter/std": 0.17855486273765564, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9357738494873047, "rewards/total_composite/std": 0.17855486273765564, "reward": 0.9357738494873047, "reward_std": 0.17855484783649445, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030108362436294556, "sampling/sampling_logp_difference/max": 7.794986724853516, "sampling/importance_sampling_ratio/min": 0.0004117942589800805, "sampling/importance_sampling_ratio/mean": 1.00041925907135, "sampling/importance_sampling_ratio/max": 1.5800968408584595, "entropy": 0.10618306044489145, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.007464349502697587, "clip_ratio/high_max": 0.007464349502697587, "clip_ratio/region_mean": 0.007464349502697587, "reward_total_mean": 0.9357738494873047, "reward_meter_mean": 0.9357738494873047, "reward_meter_std": 0.17855486273765564, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9357738494873047, "reward_total_composite_std": 0.17855486273765564} {"timestamp_utc": "2026-04-12T01:34:00Z", "mode": "train", "global_step": 2166, "epoch": 0.08699843354621038, "loss": 0.0044, "grad_norm": 2.468413829803467, "learning_rate": 3.4393939393939395e-06, "num_tokens": 4888587.0, "completions/mean_length": 116.125, "completions/min_length": 112.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.125, "completions/min_terminated_length": 112.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.996168851852417, "rewards/meter/std": 0.0008722886559553444, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9285714626312256, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.9249789118766785, "rewards/total_composite/std": 0.07558359950780869, "reward": 0.9249789118766785, "reward_std": 0.07558359950780869, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019451651722192764, "sampling/sampling_logp_difference/max": 0.8788247108459473, "sampling/importance_sampling_ratio/min": 0.41527071595191956, "sampling/importance_sampling_ratio/mean": 1.0041885375976562, "sampling/importance_sampling_ratio/max": 1.6350321769714355, "entropy": 0.13098172564059496, "clip_ratio/low_mean": 0.005332982516847551, "clip_ratio/low_min": 0.005332982516847551, "clip_ratio/high_mean": 0.008708675275556743, "clip_ratio/high_max": 0.008708675275556743, "clip_ratio/region_mean": 0.014041657792404294, "reward_total_mean": 0.9249789118766785, "reward_meter_mean": 0.996168851852417, "reward_meter_std": 0.0008722886559553444, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9285714626312256, "reward_repeat_penalty_std": 0.07636035233736038, "reward_total_composite_mean": 0.9249789118766785, "reward_total_composite_std": 0.07558359950780869} {"timestamp_utc": "2026-04-12T01:34:04Z", "mode": "train", "global_step": 2167, "epoch": 0.08703859902799534, "loss": 0.0303, "grad_norm": 7.104674816131592, "learning_rate": 3.4363636363636364e-06, "num_tokens": 4890379.0, "completions/mean_length": 73.0, "completions/min_length": 70.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.0, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9878561496734619, "rewards/meter/std": 0.010053827427327633, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9878561496734619, "rewards/total_composite/std": 0.010053827427327633, "reward": 0.9878561496734619, "reward_std": 0.010053832083940506, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.037653032690286636, "sampling/sampling_logp_difference/max": 1.5957527160644531, "sampling/importance_sampling_ratio/min": 0.2027558535337448, "sampling/importance_sampling_ratio/mean": 1.0013821125030518, "sampling/importance_sampling_ratio/max": 1.616224765777588, "entropy": 0.2611173652112484, "clip_ratio/low_mean": 0.008205128135159612, "clip_ratio/low_min": 0.008205128135159612, "clip_ratio/high_mean": 0.024240323924459517, "clip_ratio/high_max": 0.024240323924459517, "clip_ratio/region_mean": 0.03244545205961913, "reward_total_mean": 0.9878561496734619, "reward_meter_mean": 0.9878561496734619, "reward_meter_std": 0.010053827427327633, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9878561496734619, "reward_total_composite_std": 0.010053827427327633} {"timestamp_utc": "2026-04-12T01:34:08Z", "mode": "train", "global_step": 2168, "epoch": 0.08707876450978029, "loss": -0.0006, "grad_norm": 0.1801041215658188, "learning_rate": 3.4333333333333336e-06, "num_tokens": 4891883.0, "completions/mean_length": 35.0, "completions/min_length": 35.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9979922771453857, "rewards/meter/std": 1.613328822713811e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979922771453857, "rewards/total_composite/std": 1.613328822713811e-05, "reward": 0.9979922771453857, "reward_std": 1.613025233382359e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005629383493214846, "sampling/sampling_logp_difference/max": 0.34334421157836914, "sampling/importance_sampling_ratio/min": 0.7093939781188965, "sampling/importance_sampling_ratio/mean": 1.0019093751907349, "sampling/importance_sampling_ratio/max": 1.0980141162872314, "entropy": 0.051043334417045116, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0035714285913854837, "clip_ratio/high_max": 0.0035714285913854837, "clip_ratio/region_mean": 0.0035714285913854837, "reward_total_mean": 0.9979922771453857, "reward_meter_mean": 0.9979922771453857, "reward_meter_std": 1.613328822713811e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9979922771453857, "reward_total_composite_std": 1.613328822713811e-05} {"timestamp_utc": "2026-04-12T01:34:16Z", "mode": "train", "global_step": 2169, "epoch": 0.08711892999156524, "loss": 0.0195, "grad_norm": 1.7812119722366333, "learning_rate": 3.4303030303030305e-06, "num_tokens": 4896528.0, "completions/mean_length": 346.625, "completions/min_length": 336.0, "completions/max_length": 369.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 346.625, "completions/min_terminated_length": 336.0, "completions/max_terminated_length": 369.0, "rewards/meter/mean": 0.9944499731063843, "rewards/meter/std": 0.0038915579207241535, "rewards/count_adherence/mean": 0.701923131942749, "rewards/count_adherence/std": 0.027196412906050682, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8722910284996033, "rewards/repeat_penalty/std": 0.08643024414777756, "rewards/total_composite/mean": 0.6071589589118958, "rewards/total_composite/std": 0.0453052781522274, "reward": 0.6071589589118958, "reward_std": 0.045305285602808, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03724047169089317, "sampling/sampling_logp_difference/max": 5.4400858879089355, "sampling/importance_sampling_ratio/min": 0.004339110571891069, "sampling/importance_sampling_ratio/mean": 1.0060648918151855, "sampling/importance_sampling_ratio/max": 1.909399151802063, "entropy": 0.32041275314986706, "clip_ratio/low_mean": 0.006560013280250132, "clip_ratio/low_min": 0.006560013280250132, "clip_ratio/high_mean": 0.022110585356131196, "clip_ratio/high_max": 0.022110585356131196, "clip_ratio/region_mean": 0.028670598636381328, "reward_total_mean": 0.6071589589118958, "reward_meter_mean": 0.9944499731063843, "reward_meter_std": 0.0038915579207241535, "reward_count_adherence_mean": 0.701923131942749, "reward_count_adherence_std": 0.027196412906050682, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8722910284996033, "reward_repeat_penalty_std": 0.08643024414777756, "reward_total_composite_mean": 0.6071589589118958, "reward_total_composite_std": 0.0453052781522274} {"timestamp_utc": "2026-04-12T01:34:21Z", "mode": "train", "global_step": 2170, "epoch": 0.0871590954733502, "loss": -0.0018, "grad_norm": 2.5535831451416016, "learning_rate": 3.4272727272727273e-06, "num_tokens": 4898436.0, "completions/mean_length": 75.5, "completions/min_length": 73.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.5, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9986087083816528, "rewards/meter/std": 0.0005744586233049631, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9986087083816528, "rewards/total_composite/std": 0.0005744586233049631, "reward": 0.9986087083816528, "reward_std": 0.0005744565278291702, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.032794393599033356, "sampling/sampling_logp_difference/max": 1.1777338981628418, "sampling/importance_sampling_ratio/min": 0.3079758584499359, "sampling/importance_sampling_ratio/mean": 1.0089645385742188, "sampling/importance_sampling_ratio/max": 1.619355320930481, "entropy": 0.34711651876568794, "clip_ratio/low_mean": 0.008402727777138352, "clip_ratio/low_min": 0.008402727777138352, "clip_ratio/high_mean": 0.016413721488788724, "clip_ratio/high_max": 0.016413721488788724, "clip_ratio/region_mean": 0.024816449265927076, "reward_total_mean": 0.9986087083816528, "reward_meter_mean": 0.9986087083816528, "reward_meter_std": 0.0005744586233049631, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9986087083816528, "reward_total_composite_std": 0.0005744586233049631} {"timestamp_utc": "2026-04-12T01:34:26Z", "mode": "train", "global_step": 2171, "epoch": 0.08719926095513515, "loss": 0.0221, "grad_norm": 3.1598801612854004, "learning_rate": 3.4242424242424246e-06, "num_tokens": 4900337.0, "completions/mean_length": 75.625, "completions/min_length": 72.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.625, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9978922009468079, "rewards/meter/std": 0.00160902866628021, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978922009468079, "rewards/total_composite/std": 0.00160902866628021, "reward": 0.9978922009468079, "reward_std": 0.001609029364772141, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03502906486392021, "sampling/sampling_logp_difference/max": 1.1911673545837402, "sampling/importance_sampling_ratio/min": 0.30386632680892944, "sampling/importance_sampling_ratio/mean": 1.0061036348342896, "sampling/importance_sampling_ratio/max": 1.6152563095092773, "entropy": 0.3033796586096287, "clip_ratio/low_mean": 0.010196133749559522, "clip_ratio/low_min": 0.010196133749559522, "clip_ratio/high_mean": 0.018138289218768477, "clip_ratio/high_max": 0.018138289218768477, "clip_ratio/region_mean": 0.028334422968328, "reward_total_mean": 0.9978922009468079, "reward_meter_mean": 0.9978922009468079, "reward_meter_std": 0.00160902866628021, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9978922009468079, "reward_total_composite_std": 0.00160902866628021} {"timestamp_utc": "2026-04-12T01:34:32Z", "mode": "train", "global_step": 2172, "epoch": 0.0872394264369201, "loss": 0.0003, "grad_norm": 2.200758695602417, "learning_rate": 3.4212121212121214e-06, "num_tokens": 4903772.0, "completions/mean_length": 216.375, "completions/min_length": 213.0, "completions/max_length": 219.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 216.375, "completions/min_terminated_length": 213.0, "completions/max_terminated_length": 219.0, "rewards/meter/mean": 0.9980002641677856, "rewards/meter/std": 0.00031345427851192653, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9431818723678589, "rewards/repeat_penalty/std": 0.06763852387666702, "rewards/total_composite/mean": 0.8068255186080933, "rewards/total_composite/std": 0.05786222964525223, "reward": 0.8068255186080933, "reward_std": 0.057862233370542526, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028192421421408653, "sampling/sampling_logp_difference/max": 1.0793828964233398, "sampling/importance_sampling_ratio/min": 0.33980512619018555, "sampling/importance_sampling_ratio/mean": 1.0067384243011475, "sampling/importance_sampling_ratio/max": 1.921886920928955, "entropy": 0.3047393374145031, "clip_ratio/low_mean": 0.012229247367940843, "clip_ratio/low_min": 0.012229247367940843, "clip_ratio/high_mean": 0.012681030901148915, "clip_ratio/high_max": 0.012681030901148915, "clip_ratio/region_mean": 0.02491027826908976, "reward_total_mean": 0.8068255186080933, "reward_meter_mean": 0.9980002641677856, "reward_meter_std": 0.00031345427851192653, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9431818723678589, "reward_repeat_penalty_std": 0.06763852387666702, "reward_total_composite_mean": 0.8068255186080933, "reward_total_composite_std": 0.05786222964525223} {"timestamp_utc": "2026-04-12T01:34:37Z", "mode": "train", "global_step": 2173, "epoch": 0.08727959191870506, "loss": -0.0002, "grad_norm": 4.465800762176514, "learning_rate": 3.4181818181818182e-06, "num_tokens": 4905495.0, "completions/mean_length": 68.375, "completions/min_length": 68.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.375, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9475573301315308, "rewards/meter/std": 0.03694632276892662, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9475573301315308, "rewards/total_composite/std": 0.03694632276892662, "reward": 0.9475573301315308, "reward_std": 0.036946315318346024, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02550523541867733, "sampling/sampling_logp_difference/max": 0.9391798973083496, "sampling/importance_sampling_ratio/min": 0.3909483253955841, "sampling/importance_sampling_ratio/mean": 1.000171184539795, "sampling/importance_sampling_ratio/max": 1.5417495965957642, "entropy": 0.12862383667379618, "clip_ratio/low_mean": 0.0036498295376077294, "clip_ratio/low_min": 0.0036498295376077294, "clip_ratio/high_mean": 0.020063025411218405, "clip_ratio/high_max": 0.020063025411218405, "clip_ratio/region_mean": 0.023712854948826134, "reward_total_mean": 0.9475573301315308, "reward_meter_mean": 0.9475573301315308, "reward_meter_std": 0.03694632276892662, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9475573301315308, "reward_total_composite_std": 0.03694632276892662} {"timestamp_utc": "2026-04-12T01:34:42Z", "mode": "train", "global_step": 2174, "epoch": 0.08731975740049001, "loss": -0.0049, "grad_norm": 1.1670396327972412, "learning_rate": 3.415151515151515e-06, "num_tokens": 4907490.0, "completions/mean_length": 93.375, "completions/min_length": 93.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.375, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.9865900874137878, "rewards/meter/std": 0.00459433114156127, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9865900874137878, "rewards/total_composite/std": 0.00459433114156127, "reward": 0.9865900874137878, "reward_std": 0.0045943367294967175, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013781530782580376, "sampling/sampling_logp_difference/max": 2.305748224258423, "sampling/importance_sampling_ratio/min": 0.09968418627977371, "sampling/importance_sampling_ratio/mean": 0.999477744102478, "sampling/importance_sampling_ratio/max": 1.247177004814148, "entropy": 0.06566301081329584, "clip_ratio/low_mean": 0.008064516121521592, "clip_ratio/low_min": 0.008064516121521592, "clip_ratio/high_mean": 0.0013020833721384406, "clip_ratio/high_max": 0.0013020833721384406, "clip_ratio/region_mean": 0.009366599493660033, "reward_total_mean": 0.9865900874137878, "reward_meter_mean": 0.9865900874137878, "reward_meter_std": 0.00459433114156127, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9865900874137878, "reward_total_composite_std": 0.00459433114156127} {"timestamp_utc": "2026-04-12T01:34:46Z", "mode": "train", "global_step": 2175, "epoch": 0.08735992288227497, "loss": -0.0057, "grad_norm": 2.544595956802368, "learning_rate": 3.4121212121212123e-06, "num_tokens": 4909423.0, "completions/mean_length": 85.625, "completions/min_length": 84.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.625, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.9955015778541565, "rewards/meter/std": 0.00046232930617406964, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9955015778541565, "rewards/total_composite/std": 0.00046232930617406964, "reward": 0.9955015778541565, "reward_std": 0.00046233335160650313, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009426219388842583, "sampling/sampling_logp_difference/max": 0.2666025161743164, "sampling/importance_sampling_ratio/min": 0.7659775018692017, "sampling/importance_sampling_ratio/mean": 1.005947232246399, "sampling/importance_sampling_ratio/max": 1.216043472290039, "entropy": 0.09059183578938246, "clip_ratio/low_mean": 0.0029411765281111, "clip_ratio/low_min": 0.0029411765281111, "clip_ratio/high_mean": 0.0058139534667134285, "clip_ratio/high_max": 0.0058139534667134285, "clip_ratio/region_mean": 0.008755129994824529, "reward_total_mean": 0.9955015778541565, "reward_meter_mean": 0.9955015778541565, "reward_meter_std": 0.00046232930617406964, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9955015778541565, "reward_total_composite_std": 0.00046232930617406964} {"timestamp_utc": "2026-04-12T01:34:51Z", "mode": "train", "global_step": 2176, "epoch": 0.08740008836405992, "loss": -0.001, "grad_norm": 2.064898729324341, "learning_rate": 3.409090909090909e-06, "num_tokens": 4911299.0, "completions/mean_length": 80.5, "completions/min_length": 80.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.5, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.7326533794403076, "rewards/meter/std": 0.055831361562013626, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.69842928647995, "rewards/total_composite/std": 0.10409945249557495, "reward": 0.69842928647995, "reward_std": 0.10409945994615555, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008284451439976692, "sampling/sampling_logp_difference/max": 0.7730828523635864, "sampling/importance_sampling_ratio/min": 0.4615878760814667, "sampling/importance_sampling_ratio/mean": 0.9991859197616577, "sampling/importance_sampling_ratio/max": 1.3141322135925293, "entropy": 0.035707658622413874, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.006137048127129674, "clip_ratio/high_max": 0.006137048127129674, "clip_ratio/region_mean": 0.006137048127129674, "reward_total_mean": 0.69842928647995, "reward_meter_mean": 0.7326533794403076, "reward_meter_std": 0.055831361562013626, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.69842928647995, "reward_total_composite_std": 0.10409945249557495} {"timestamp_utc": "2026-04-12T01:34:55Z", "mode": "train", "global_step": 2177, "epoch": 0.08744025384584488, "loss": -0.0033, "grad_norm": 0.053646668791770935, "learning_rate": 3.406060606060606e-06, "num_tokens": 4912899.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.7176748514175415, "rewards/meter/std": 0.19788797199726105, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7176748514175415, "rewards/total_composite/std": 0.19788797199726105, "reward": 0.7176748514175415, "reward_std": 0.19788797199726105, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00239483080804348, "sampling/sampling_logp_difference/max": 0.5967006683349609, "sampling/importance_sampling_ratio/min": 0.5506253242492676, "sampling/importance_sampling_ratio/mean": 0.999964714050293, "sampling/importance_sampling_ratio/max": 1.0113025903701782, "entropy": 0.009889230597764254, "clip_ratio/low_mean": 0.002314814832061529, "clip_ratio/low_min": 0.002314814832061529, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.002314814832061529, "reward_total_mean": 0.7176748514175415, "reward_meter_mean": 0.7176748514175415, "reward_meter_std": 0.19788797199726105, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7176748514175415, "reward_total_composite_std": 0.19788797199726105} {"timestamp_utc": "2026-04-12T01:35:01Z", "mode": "train", "global_step": 2178, "epoch": 0.08748041932762983, "loss": 0.0009, "grad_norm": 1.710945725440979, "learning_rate": 3.4030303030303036e-06, "num_tokens": 4915728.0, "completions/mean_length": 170.625, "completions/min_length": 170.0, "completions/max_length": 171.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 170.625, "completions/min_terminated_length": 170.0, "completions/max_terminated_length": 171.0, "rewards/meter/mean": 0.9990792274475098, "rewards/meter/std": 7.88669494795613e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9166666269302368, "rewards/repeat_penalty/std": 0.05143444612622261, "rewards/total_composite/mean": 0.9158225059509277, "rewards/total_composite/std": 0.05138414725661278, "reward": 0.9158225059509277, "reward_std": 0.05138415843248367, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01991666667163372, "sampling/sampling_logp_difference/max": 1.5503005981445312, "sampling/importance_sampling_ratio/min": 0.3940596580505371, "sampling/importance_sampling_ratio/mean": 1.0056817531585693, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1335141584277153, "clip_ratio/low_mean": 0.00733574153855443, "clip_ratio/low_min": 0.00733574153855443, "clip_ratio/high_mean": 0.0021929824724793434, "clip_ratio/high_max": 0.0021929824724793434, "clip_ratio/region_mean": 0.009528724011033773, "reward_total_mean": 0.9158225059509277, "reward_meter_mean": 0.9990792274475098, "reward_meter_std": 7.88669494795613e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9166666269302368, "reward_repeat_penalty_std": 0.05143444612622261, "reward_total_composite_mean": 0.9158225059509277, "reward_total_composite_std": 0.05138414725661278} {"timestamp_utc": "2026-04-12T01:35:06Z", "mode": "train", "global_step": 2179, "epoch": 0.08752058480941478, "loss": 0.0006, "grad_norm": 0.20853394269943237, "learning_rate": 3.4000000000000005e-06, "num_tokens": 4917552.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9990785121917725, "rewards/meter/std": 9.964550372387748e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990785121917725, "rewards/total_composite/std": 9.964550372387748e-06, "reward": 0.9990785121917725, "reward_std": 9.972212865250185e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0065149967558681965, "sampling/sampling_logp_difference/max": 0.41229283809661865, "sampling/importance_sampling_ratio/min": 0.6621303558349609, "sampling/importance_sampling_ratio/mean": 1.002784252166748, "sampling/importance_sampling_ratio/max": 1.365889072418213, "entropy": 0.05868239142000675, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/high_mean": 0.0037878789007663727, "clip_ratio/high_max": 0.0037878789007663727, "clip_ratio/region_mean": 0.005681818351149559, "reward_total_mean": 0.9990785121917725, "reward_meter_mean": 0.9990785121917725, "reward_meter_std": 9.964550372387748e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990785121917725, "reward_total_composite_std": 9.964550372387748e-06} {"timestamp_utc": "2026-04-12T01:35:13Z", "mode": "train", "global_step": 2180, "epoch": 0.08756075029119974, "loss": -0.0137, "grad_norm": 2.8113794326782227, "learning_rate": 3.3969696969696973e-06, "num_tokens": 4921059.0, "completions/mean_length": 219.375, "completions/min_length": 210.0, "completions/max_length": 236.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 219.375, "completions/min_terminated_length": 210.0, "completions/max_terminated_length": 236.0, "rewards/meter/mean": 0.8746868968009949, "rewards/meter/std": 0.34104546904563904, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9318181872367859, "rewards/repeat_penalty/std": 0.08058229833841324, "rewards/total_composite/mean": 0.7102342844009399, "rewards/total_composite/std": 0.2835994064807892, "reward": 0.7102342844009399, "reward_std": 0.2835994064807892, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.033945005387067795, "sampling/sampling_logp_difference/max": 1.3039984703063965, "sampling/importance_sampling_ratio/min": 0.27144426107406616, "sampling/importance_sampling_ratio/mean": 1.002653956413269, "sampling/importance_sampling_ratio/max": 1.8379487991333008, "entropy": 0.3587251305580139, "clip_ratio/low_mean": 0.005902778008021414, "clip_ratio/low_min": 0.005902778008021414, "clip_ratio/high_mean": 0.02744697011075914, "clip_ratio/high_max": 0.02744697011075914, "clip_ratio/region_mean": 0.03334974811878055, "reward_total_mean": 0.7102342844009399, "reward_meter_mean": 0.8746868968009949, "reward_meter_std": 0.34104546904563904, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9318181872367859, "reward_repeat_penalty_std": 0.08058229833841324, "reward_total_composite_mean": 0.7102342844009399, "reward_total_composite_std": 0.2835994064807892} {"timestamp_utc": "2026-04-12T01:35:18Z", "mode": "train", "global_step": 2181, "epoch": 0.08760091577298469, "loss": 0.0015, "grad_norm": 2.3878560066223145, "learning_rate": 3.3939393939393946e-06, "num_tokens": 4923237.0, "completions/mean_length": 97.25, "completions/min_length": 97.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.25, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.8764194250106812, "rewards/meter/std": 0.24076080322265625, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8764194250106812, "rewards/total_composite/std": 0.24076080322265625, "reward": 0.8764194250106812, "reward_std": 0.24076080322265625, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012246760539710522, "sampling/sampling_logp_difference/max": 1.104780673980713, "sampling/importance_sampling_ratio/min": 0.3312835097312927, "sampling/importance_sampling_ratio/mean": 1.0025182962417603, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0856907800771296, "clip_ratio/low_mean": 0.0025641699321568012, "clip_ratio/low_min": 0.0025641699321568012, "clip_ratio/high_mean": 0.0064301491947844625, "clip_ratio/high_max": 0.0064301491947844625, "clip_ratio/region_mean": 0.008994319126941264, "reward_total_mean": 0.8764194250106812, "reward_meter_mean": 0.8764194250106812, "reward_meter_std": 0.24076080322265625, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8764194250106812, "reward_total_composite_std": 0.24076080322265625} {"timestamp_utc": "2026-04-12T01:35:24Z", "mode": "train", "global_step": 2182, "epoch": 0.08764108125476965, "loss": -0.0054, "grad_norm": 2.051013946533203, "learning_rate": 3.3909090909090914e-06, "num_tokens": 4925782.0, "completions/mean_length": 127.125, "completions/min_length": 125.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.125, "completions/min_terminated_length": 125.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.9971377849578857, "rewards/meter/std": 0.0010128796566277742, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8546895980834961, "rewards/total_composite/std": 0.0008681902545504272, "reward": 0.8546895980834961, "reward_std": 0.0008681949693709612, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011722107417881489, "sampling/sampling_logp_difference/max": 1.6086368560791016, "sampling/importance_sampling_ratio/min": 0.20016026496887207, "sampling/importance_sampling_ratio/mean": 1.0007635354995728, "sampling/importance_sampling_ratio/max": 1.307826042175293, "entropy": 0.07762610260397196, "clip_ratio/low_mean": 0.001953125, "clip_ratio/low_min": 0.001953125, "clip_ratio/high_mean": 0.005906250094994903, "clip_ratio/high_max": 0.005906250094994903, "clip_ratio/region_mean": 0.007859375094994903, "reward_total_mean": 0.8546895980834961, "reward_meter_mean": 0.9971377849578857, "reward_meter_std": 0.0010128796566277742, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8546895980834961, "reward_total_composite_std": 0.0008681902545504272} {"timestamp_utc": "2026-04-12T01:35:29Z", "mode": "train", "global_step": 2183, "epoch": 0.0876812467365546, "loss": 0.0002, "grad_norm": 3.130744457244873, "learning_rate": 3.3878787878787882e-06, "num_tokens": 4927521.0, "completions/mean_length": 66.375, "completions/min_length": 66.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.375, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9979628324508667, "rewards/meter/std": 0.0002601691521704197, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979628324508667, "rewards/total_composite/std": 0.0002601691521704197, "reward": 0.9979628324508667, "reward_std": 0.0002601697633508593, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010011628270149231, "sampling/sampling_logp_difference/max": 0.9755148887634277, "sampling/importance_sampling_ratio/min": 0.3769982159137726, "sampling/importance_sampling_ratio/mean": 1.00442373752594, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07284869812428951, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.005599473137408495, "clip_ratio/high_max": 0.005599473137408495, "clip_ratio/region_mean": 0.005599473137408495, "reward_total_mean": 0.9979628324508667, "reward_meter_mean": 0.9979628324508667, "reward_meter_std": 0.0002601691521704197, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9979628324508667, "reward_total_composite_std": 0.0002601691521704197} {"timestamp_utc": "2026-04-12T01:35:35Z", "mode": "train", "global_step": 2184, "epoch": 0.08772141221833955, "loss": 0.0252, "grad_norm": 4.508577346801758, "learning_rate": 3.384848484848485e-06, "num_tokens": 4929678.0, "completions/mean_length": 110.625, "completions/min_length": 106.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 110.625, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.8917012214660645, "rewards/meter/std": 0.3005649745464325, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8667252063751221, "rewards/total_composite/std": 0.29866674542427063, "reward": 0.8667252063751221, "reward_std": 0.29866674542427063, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05617932230234146, "sampling/sampling_logp_difference/max": 1.6177425384521484, "sampling/importance_sampling_ratio/min": 0.19834595918655396, "sampling/importance_sampling_ratio/mean": 1.0148777961730957, "sampling/importance_sampling_ratio/max": 1.945900797843933, "entropy": 0.5080476962029934, "clip_ratio/low_mean": 0.014024864183738828, "clip_ratio/low_min": 0.014024864183738828, "clip_ratio/high_mean": 0.031617405358701944, "clip_ratio/high_max": 0.031617405358701944, "clip_ratio/region_mean": 0.04564226954244077, "reward_total_mean": 0.8667252063751221, "reward_meter_mean": 0.8917012214660645, "reward_meter_std": 0.3005649745464325, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.8667252063751221, "reward_total_composite_std": 0.29866674542427063} {"timestamp_utc": "2026-04-12T01:35:40Z", "mode": "train", "global_step": 2185, "epoch": 0.08776157770012451, "loss": 0.001, "grad_norm": 3.390443801879883, "learning_rate": 3.3818181818181823e-06, "num_tokens": 4931678.0, "completions/mean_length": 80.0, "completions/min_length": 80.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.0, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.7347736954689026, "rewards/meter/std": 0.02520820125937462, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.6960331201553345, "rewards/total_composite/std": 0.047284774482250214, "reward": 0.6960331201553345, "reward_std": 0.04728476703166962, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0048207431100308895, "sampling/sampling_logp_difference/max": 0.8722677230834961, "sampling/importance_sampling_ratio/min": 0.41800257563591003, "sampling/importance_sampling_ratio/mean": 1.000468134880066, "sampling/importance_sampling_ratio/max": 1.1883066892623901, "entropy": 0.02709749317727983, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0015625000232830644, "clip_ratio/high_max": 0.0015625000232830644, "clip_ratio/region_mean": 0.0015625000232830644, "reward_total_mean": 0.6960331201553345, "reward_meter_mean": 0.7347736954689026, "reward_meter_std": 0.02520820125937462, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.6960331201553345, "reward_total_composite_std": 0.047284774482250214} {"timestamp_utc": "2026-04-12T01:35:45Z", "mode": "train", "global_step": 2186, "epoch": 0.08780174318190946, "loss": -0.0159, "grad_norm": 4.1797590255737305, "learning_rate": 3.378787878787879e-06, "num_tokens": 4933511.0, "completions/mean_length": 75.125, "completions/min_length": 70.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.125, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9933065176010132, "rewards/meter/std": 0.01265679020434618, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9933065176010132, "rewards/total_composite/std": 0.01265679020434618, "reward": 0.9933065176010132, "reward_std": 0.012656783685088158, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03881411254405975, "sampling/sampling_logp_difference/max": 1.1373529434204102, "sampling/importance_sampling_ratio/min": 0.32066673040390015, "sampling/importance_sampling_ratio/mean": 1.0102565288543701, "sampling/importance_sampling_ratio/max": 1.9696723222732544, "entropy": 0.3504325784742832, "clip_ratio/low_mean": 0.0017857142956927419, "clip_ratio/low_min": 0.0017857142956927419, "clip_ratio/high_mean": 0.036382337333634496, "clip_ratio/high_max": 0.036382337333634496, "clip_ratio/region_mean": 0.03816805162932724, "reward_total_mean": 0.9933065176010132, "reward_meter_mean": 0.9933065176010132, "reward_meter_std": 0.01265679020434618, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9933065176010132, "reward_total_composite_std": 0.01265679020434618} {"timestamp_utc": "2026-04-12T01:35:53Z", "mode": "train", "global_step": 2187, "epoch": 0.08784190866369442, "loss": -0.0221, "grad_norm": 1.9298547506332397, "learning_rate": 3.375757575757576e-06, "num_tokens": 4938469.0, "completions/mean_length": 314.75, "completions/min_length": 297.0, "completions/max_length": 354.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 314.75, "completions/min_terminated_length": 297.0, "completions/max_terminated_length": 354.0, "rewards/meter/mean": 0.997092604637146, "rewards/meter/std": 0.0013639559037983418, "rewards/count_adherence/mean": 0.5803571939468384, "rewards/count_adherence/std": 0.02525380253791809, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9509804248809814, "rewards/repeat_penalty/std": 0.04682480916380882, "rewards/total_composite/mean": 0.5502004623413086, "rewards/total_composite/std": 0.03388208523392677, "reward": 0.5502004623413086, "reward_std": 0.03388208895921707, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04348405450582504, "sampling/sampling_logp_difference/max": 1.2296199798583984, "sampling/importance_sampling_ratio/min": 0.2924036681652069, "sampling/importance_sampling_ratio/mean": 1.0096619129180908, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45917268842458725, "clip_ratio/low_mean": 0.015058388700708747, "clip_ratio/low_min": 0.015058388700708747, "clip_ratio/high_mean": 0.02572961524128914, "clip_ratio/high_max": 0.02572961524128914, "clip_ratio/region_mean": 0.040788003941997886, "reward_total_mean": 0.5502004623413086, "reward_meter_mean": 0.997092604637146, "reward_meter_std": 0.0013639559037983418, "reward_count_adherence_mean": 0.5803571939468384, "reward_count_adherence_std": 0.02525380253791809, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9509804248809814, "reward_repeat_penalty_std": 0.04682480916380882, "reward_total_composite_mean": 0.5502004623413086, "reward_total_composite_std": 0.03388208523392677} {"timestamp_utc": "2026-04-12T01:35:58Z", "mode": "train", "global_step": 2188, "epoch": 0.08788207414547937, "loss": 0.0088, "grad_norm": 1.814806342124939, "learning_rate": 3.3727272727272732e-06, "num_tokens": 4940773.0, "completions/mean_length": 115.0, "completions/min_length": 113.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.0, "completions/min_terminated_length": 113.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.9959709644317627, "rewards/meter/std": 0.00034307397436350584, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.910714328289032, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9070344567298889, "rewards/total_composite/std": 0.07348322868347168, "reward": 0.9070344567298889, "reward_std": 0.07348321378231049, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015008204616606236, "sampling/sampling_logp_difference/max": 1.293868064880371, "sampling/importance_sampling_ratio/min": 0.27420806884765625, "sampling/importance_sampling_ratio/mean": 0.9997696280479431, "sampling/importance_sampling_ratio/max": 1.2936325073242188, "entropy": 0.07655041664838791, "clip_ratio/low_mean": 0.0010775862028822303, "clip_ratio/low_min": 0.0010775862028822303, "clip_ratio/high_mean": 0.006579951150342822, "clip_ratio/high_max": 0.006579951150342822, "clip_ratio/region_mean": 0.007657537353225052, "reward_total_mean": 0.9070344567298889, "reward_meter_mean": 0.9959709644317627, "reward_meter_std": 0.00034307397436350584, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.910714328289032, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.9070344567298889, "reward_total_composite_std": 0.07348322868347168} {"timestamp_utc": "2026-04-12T01:36:03Z", "mode": "train", "global_step": 2189, "epoch": 0.08792223962726432, "loss": 0.0002, "grad_norm": 0.7317746877670288, "learning_rate": 3.36969696969697e-06, "num_tokens": 4942493.0, "completions/mean_length": 56.0, "completions/min_length": 56.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.0, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.9949744939804077, "rewards/meter/std": 3.23687017953489e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9949744939804077, "rewards/total_composite/std": 3.23687017953489e-05, "reward": 0.9949744939804077, "reward_std": 3.236968768760562e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00435090996325016, "sampling/sampling_logp_difference/max": 0.32142794132232666, "sampling/importance_sampling_ratio/min": 0.8439798355102539, "sampling/importance_sampling_ratio/mean": 1.003525972366333, "sampling/importance_sampling_ratio/max": 1.3790956735610962, "entropy": 0.05346089042723179, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0022321429569274187, "clip_ratio/high_max": 0.0022321429569274187, "clip_ratio/region_mean": 0.0022321429569274187, "reward_total_mean": 0.9949744939804077, "reward_meter_mean": 0.9949744939804077, "reward_meter_std": 3.23687017953489e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9949744939804077, "reward_total_composite_std": 3.23687017953489e-05} {"timestamp_utc": "2026-04-12T01:36:07Z", "mode": "train", "global_step": 2190, "epoch": 0.08796240510904928, "loss": 0.0157, "grad_norm": 3.3835291862487793, "learning_rate": 3.366666666666667e-06, "num_tokens": 4943784.0, "completions/mean_length": 33.375, "completions/min_length": 33.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.375, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9737765789031982, "rewards/meter/std": 0.008460394106805325, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9737765789031982, "rewards/total_composite/std": 0.008460394106805325, "reward": 0.9737765789031982, "reward_std": 0.008460381999611855, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019433647394180298, "sampling/sampling_logp_difference/max": 1.1026906967163086, "sampling/importance_sampling_ratio/min": 0.33197662234306335, "sampling/importance_sampling_ratio/mean": 1.0068951845169067, "sampling/importance_sampling_ratio/max": 1.51780366897583, "entropy": 0.14891941286623478, "clip_ratio/low_mean": 0.011140820104628801, "clip_ratio/low_min": 0.011140820104628801, "clip_ratio/high_mean": 0.0037878789007663727, "clip_ratio/high_max": 0.0037878789007663727, "clip_ratio/region_mean": 0.014928699005395174, "reward_total_mean": 0.9737765789031982, "reward_meter_mean": 0.9737765789031982, "reward_meter_std": 0.008460394106805325, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9737765789031982, "reward_total_composite_std": 0.008460394106805325} {"timestamp_utc": "2026-04-12T01:36:13Z", "mode": "train", "global_step": 2191, "epoch": 0.08800257059083423, "loss": 0.0003, "grad_norm": 1.943103551864624, "learning_rate": 3.3636363636363637e-06, "num_tokens": 4946182.0, "completions/mean_length": 113.75, "completions/min_length": 112.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 113.75, "completions/min_terminated_length": 112.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.995752215385437, "rewards/meter/std": 0.00029232812812551856, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.9601929187774658, "rewards/total_composite/std": 0.06590110063552856, "reward": 0.9601929187774658, "reward_std": 0.06590110063552856, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015357456170022488, "sampling/sampling_logp_difference/max": 1.5764732360839844, "sampling/importance_sampling_ratio/min": 0.2067027986049652, "sampling/importance_sampling_ratio/mean": 1.0013221502304077, "sampling/importance_sampling_ratio/max": 1.6537041664123535, "entropy": 0.09450818877667189, "clip_ratio/low_mean": 0.0021739129442721605, "clip_ratio/low_min": 0.0021739129442721605, "clip_ratio/high_mean": 0.006637688144110143, "clip_ratio/high_max": 0.006637688144110143, "clip_ratio/region_mean": 0.008811601088382304, "reward_total_mean": 0.9601929187774658, "reward_meter_mean": 0.995752215385437, "reward_meter_std": 0.00029232812812551856, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.9601929187774658, "reward_total_composite_std": 0.06590110063552856} {"timestamp_utc": "2026-04-12T01:36:18Z", "mode": "train", "global_step": 2192, "epoch": 0.08804273607261918, "loss": 0.0058, "grad_norm": 3.4850220680236816, "learning_rate": 3.360606060606061e-06, "num_tokens": 4948311.0, "completions/mean_length": 100.125, "completions/min_length": 99.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.125, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.99900221824646, "rewards/meter/std": 0.00022944888041820377, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.99900221824646, "rewards/total_composite/std": 0.00022944888041820377, "reward": 0.99900221824646, "reward_std": 0.00022944044030737132, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01750640757381916, "sampling/sampling_logp_difference/max": 1.2861604690551758, "sampling/importance_sampling_ratio/min": 0.27632972598075867, "sampling/importance_sampling_ratio/mean": 0.9979958534240723, "sampling/importance_sampling_ratio/max": 1.8097566366195679, "entropy": 0.08774240221828222, "clip_ratio/low_mean": 0.0012376237427815795, "clip_ratio/low_min": 0.0012376237427815795, "clip_ratio/high_mean": 0.018826007028110325, "clip_ratio/high_max": 0.018826007028110325, "clip_ratio/region_mean": 0.020063630770891905, "reward_total_mean": 0.99900221824646, "reward_meter_mean": 0.99900221824646, "reward_meter_std": 0.00022944888041820377, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.99900221824646, "reward_total_composite_std": 0.00022944888041820377} {"timestamp_utc": "2026-04-12T01:36:23Z", "mode": "train", "global_step": 2193, "epoch": 0.08808290155440414, "loss": 0.0298, "grad_norm": 5.769536972045898, "learning_rate": 3.357575757575758e-06, "num_tokens": 4950310.0, "completions/mean_length": 75.875, "completions/min_length": 72.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.875, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.9790392518043518, "rewards/meter/std": 0.02525605633854866, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9790392518043518, "rewards/total_composite/std": 0.02525605633854866, "reward": 0.9790392518043518, "reward_std": 0.025256047025322914, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06429249048233032, "sampling/sampling_logp_difference/max": 1.042258858680725, "sampling/importance_sampling_ratio/min": 0.35265716910362244, "sampling/importance_sampling_ratio/mean": 1.0126549005508423, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5629463791847229, "clip_ratio/low_mean": 0.031126924557611346, "clip_ratio/low_min": 0.031126924557611346, "clip_ratio/high_mean": 0.014968461007811129, "clip_ratio/high_max": 0.014968461007811129, "clip_ratio/region_mean": 0.046095385565422475, "reward_total_mean": 0.9790392518043518, "reward_meter_mean": 0.9790392518043518, "reward_meter_std": 0.02525605633854866, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9790392518043518, "reward_total_composite_std": 0.02525605633854866} {"timestamp_utc": "2026-04-12T01:36:28Z", "mode": "train", "global_step": 2194, "epoch": 0.08812306703618909, "loss": 0.0016, "grad_norm": 4.066121578216553, "learning_rate": 3.3545454545454547e-06, "num_tokens": 4952168.0, "completions/mean_length": 69.25, "completions/min_length": 69.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.25, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9961166977882385, "rewards/meter/std": 0.0030905893072485924, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9961166977882385, "rewards/total_composite/std": 0.0030905893072485924, "reward": 0.9961166977882385, "reward_std": 0.0030905804596841335, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008842971175909042, "sampling/sampling_logp_difference/max": 0.6939091682434082, "sampling/importance_sampling_ratio/min": 0.49961915612220764, "sampling/importance_sampling_ratio/mean": 0.9982797503471375, "sampling/importance_sampling_ratio/max": 1.0906788110733032, "entropy": 0.04704178823158145, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.003597308532334864, "clip_ratio/high_max": 0.003597308532334864, "clip_ratio/region_mean": 0.003597308532334864, "reward_total_mean": 0.9961166977882385, "reward_meter_mean": 0.9961166977882385, "reward_meter_std": 0.0030905893072485924, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9961166977882385, "reward_total_composite_std": 0.0030905893072485924} {"timestamp_utc": "2026-04-12T01:36:33Z", "mode": "train", "global_step": 2195, "epoch": 0.08816323251797405, "loss": 0.0008, "grad_norm": 3.4402453899383545, "learning_rate": 3.351515151515152e-06, "num_tokens": 4953959.0, "completions/mean_length": 75.875, "completions/min_length": 72.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.875, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.998624861240387, "rewards/meter/std": 0.0006542301853187382, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998624861240387, "rewards/total_composite/std": 0.0006542301853187382, "reward": 0.998624861240387, "reward_std": 0.0006542490446008742, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04149501770734787, "sampling/sampling_logp_difference/max": 1.0772240161895752, "sampling/importance_sampling_ratio/min": 0.3405395448207855, "sampling/importance_sampling_ratio/mean": 1.0039963722229004, "sampling/importance_sampling_ratio/max": 1.4867199659347534, "entropy": 0.3429943472146988, "clip_ratio/low_mean": 0.00645240826997906, "clip_ratio/low_min": 0.00645240826997906, "clip_ratio/high_mean": 0.013268729439005256, "clip_ratio/high_max": 0.013268729439005256, "clip_ratio/region_mean": 0.019721137708984315, "reward_total_mean": 0.998624861240387, "reward_meter_mean": 0.998624861240387, "reward_meter_std": 0.0006542301853187382, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.998624861240387, "reward_total_composite_std": 0.0006542301853187382} {"timestamp_utc": "2026-04-12T01:36:37Z", "mode": "train", "global_step": 2196, "epoch": 0.088203397999759, "loss": 0.0022, "grad_norm": 1.1731183528900146, "learning_rate": 3.3484848484848487e-06, "num_tokens": 4955804.0, "completions/mean_length": 65.625, "completions/min_length": 65.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.625, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9991217255592346, "rewards/meter/std": 3.6965378967579454e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991217255592346, "rewards/total_composite/std": 3.6965378967579454e-05, "reward": 0.9991217255592346, "reward_std": 3.6982291931053624e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007183389738202095, "sampling/sampling_logp_difference/max": 0.5108299255371094, "sampling/importance_sampling_ratio/min": 0.7030104994773865, "sampling/importance_sampling_ratio/mean": 1.0017669200897217, "sampling/importance_sampling_ratio/max": 1.6666737794876099, "entropy": 0.04586625238880515, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.003846153849735856, "clip_ratio/high_max": 0.003846153849735856, "clip_ratio/region_mean": 0.003846153849735856, "reward_total_mean": 0.9991217255592346, "reward_meter_mean": 0.9991217255592346, "reward_meter_std": 3.6965378967579454e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9991217255592346, "reward_total_composite_std": 3.6965378967579454e-05} {"timestamp_utc": "2026-04-12T01:36:42Z", "mode": "train", "global_step": 2197, "epoch": 0.08824356348154395, "loss": 0.0579, "grad_norm": 9.319520950317383, "learning_rate": 3.3454545454545456e-06, "num_tokens": 4957640.0, "completions/mean_length": 75.5, "completions/min_length": 71.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.5, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9673671126365662, "rewards/meter/std": 0.03739238530397415, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9673671126365662, "rewards/total_composite/std": 0.03739238530397415, "reward": 0.9673671126365662, "reward_std": 0.03739236667752266, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.032729994505643845, "sampling/sampling_logp_difference/max": 1.7551674842834473, "sampling/importance_sampling_ratio/min": 0.17287828028202057, "sampling/importance_sampling_ratio/mean": 1.0026880502700806, "sampling/importance_sampling_ratio/max": 1.8825844526290894, "entropy": 0.20823358744382858, "clip_ratio/low_mean": 0.010880606481805444, "clip_ratio/low_min": 0.010880606481805444, "clip_ratio/high_mean": 0.011807307717390358, "clip_ratio/high_max": 0.011807307717390358, "clip_ratio/region_mean": 0.022687914199195802, "reward_total_mean": 0.9673671126365662, "reward_meter_mean": 0.9673671126365662, "reward_meter_std": 0.03739238530397415, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9673671126365662, "reward_total_composite_std": 0.03739238530397415} {"timestamp_utc": "2026-04-12T01:36:47Z", "mode": "train", "global_step": 2198, "epoch": 0.08828372896332891, "loss": 0.0073, "grad_norm": 3.934696912765503, "learning_rate": 3.3424242424242424e-06, "num_tokens": 4959602.0, "completions/mean_length": 75.25, "completions/min_length": 72.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.25, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9990965127944946, "rewards/meter/std": 0.00029607766191475093, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990965127944946, "rewards/total_composite/std": 0.00029607766191475093, "reward": 0.9990965127944946, "reward_std": 0.00029608141630887985, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021792318671941757, "sampling/sampling_logp_difference/max": 0.8232808113098145, "sampling/importance_sampling_ratio/min": 0.4389890432357788, "sampling/importance_sampling_ratio/mean": 1.0073840618133545, "sampling/importance_sampling_ratio/max": 1.3668806552886963, "entropy": 0.24795558489859104, "clip_ratio/low_mean": 0.0032051282469183207, "clip_ratio/low_min": 0.0032051282469183207, "clip_ratio/high_mean": 0.006831709994003177, "clip_ratio/high_max": 0.006831709994003177, "clip_ratio/region_mean": 0.010036838240921497, "reward_total_mean": 0.9990965127944946, "reward_meter_mean": 0.9990965127944946, "reward_meter_std": 0.00029607766191475093, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990965127944946, "reward_total_composite_std": 0.00029607766191475093} {"timestamp_utc": "2026-04-12T01:36:52Z", "mode": "train", "global_step": 2199, "epoch": 0.08832389444511386, "loss": 0.0001, "grad_norm": 0.02073613740503788, "learning_rate": 3.3393939393939397e-06, "num_tokens": 4961586.0, "completions/mean_length": 69.0, "completions/min_length": 69.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.0, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9973071813583374, "rewards/meter/std": 2.9068671665299917e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973071813583374, "rewards/total_composite/std": 2.9068671665299917e-06, "reward": 0.9973071813583374, "reward_std": 2.895276793424273e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0027781545650213957, "sampling/sampling_logp_difference/max": 0.15232563018798828, "sampling/importance_sampling_ratio/min": 0.8587086200714111, "sampling/importance_sampling_ratio/mean": 1.0010336637496948, "sampling/importance_sampling_ratio/max": 1.067936897277832, "entropy": 0.025351210730150342, "clip_ratio/low_mean": 0.0036231884732842445, "clip_ratio/low_min": 0.0036231884732842445, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0036231884732842445, "reward_total_mean": 0.9973071813583374, "reward_meter_mean": 0.9973071813583374, "reward_meter_std": 2.9068671665299917e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9973071813583374, "reward_total_composite_std": 2.9068671665299917e-06} {"timestamp_utc": "2026-04-12T01:36:56Z", "mode": "train", "global_step": 2200, "epoch": 0.08836405992689882, "loss": -0.0049, "grad_norm": 3.2771596908569336, "learning_rate": 3.3363636363636365e-06, "num_tokens": 4963619.0, "completions/mean_length": 80.125, "completions/min_length": 77.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.125, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.9969666600227356, "rewards/meter/std": 0.001407405361533165, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8722573518753052, "rewards/total_composite/std": 0.35244786739349365, "reward": 0.8722573518753052, "reward_std": 0.35244786739349365, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.038304030895233154, "sampling/sampling_logp_difference/max": 1.204660415649414, "sampling/importance_sampling_ratio/min": 0.2997937798500061, "sampling/importance_sampling_ratio/mean": 1.0073878765106201, "sampling/importance_sampling_ratio/max": 1.6984498500823975, "entropy": 0.3972342722117901, "clip_ratio/low_mean": 0.0031645570416003466, "clip_ratio/low_min": 0.0031645570416003466, "clip_ratio/high_mean": 0.03107945283409208, "clip_ratio/high_max": 0.03107945283409208, "clip_ratio/region_mean": 0.03424400987569243, "reward_total_mean": 0.8722573518753052, "reward_meter_mean": 0.9969666600227356, "reward_meter_std": 0.001407405361533165, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8722573518753052, "reward_total_composite_std": 0.35244786739349365} {"timestamp_utc": "2026-04-12T01:38:02Z", "mode": "eval", "global_step": 2200, "epoch": 0.08836405992689882, "eval_loss": NaN, "eval_runtime": 65.1539, "eval_samples_per_second": 1.596, "eval_steps_per_second": 0.2, "eval_num_tokens": 4963619.0, "eval_completions/mean_length": 187.77884615384616, "eval_completions/min_length": 62.69230769230769, "eval_completions/max_length": 339.15384615384613, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 187.77884615384616, "eval_completions/min_terminated_length": 62.69230769230769, "eval_completions/max_terminated_length": 339.15384615384613, "eval_rewards/meter/mean": 0.722443516437824, "eval_rewards/meter/std": 0.41069405812483567, "eval_rewards/count_adherence/mean": 0.8735798276387728, "eval_rewards/count_adherence/std": 0.14306257034723574, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/repeat_penalty/mean": 0.8453576381389911, "eval_rewards/repeat_penalty/std": 0.13899515225337103, "eval_rewards/total_composite/mean": 0.5540328117517325, "eval_rewards/total_composite/std": 0.3618074002174231, "eval_reward": 0.5540328117517325, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.01593467600357074, "eval_sampling/sampling_logp_difference/max": 1.04065675001878, "eval_sampling/importance_sampling_ratio/min": 0.3679314060853078, "eval_sampling/importance_sampling_ratio/mean": 1.0049138986147368, "eval_sampling/importance_sampling_ratio/max": 1.4119950441213756, "eval_entropy": 0.1745165208211312, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5540328117517325, "eval_reward_meter_mean": 0.722443516437824, "eval_reward_meter_std": 0.41069405812483567, "eval_reward_count_adherence_mean": 0.8735798276387728, "eval_reward_count_adherence_std": 0.14306257034723574, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_repeat_penalty_mean": 0.8453576381389911, "eval_reward_repeat_penalty_std": 0.13899515225337103, "eval_reward_total_composite_mean": 0.5540328117517325, "eval_reward_total_composite_std": 0.3618074002174231} {"timestamp_utc": "2026-04-12T01:38:09Z", "mode": "train", "global_step": 2201, "epoch": 0.08840422540868377, "loss": 0.0037, "grad_norm": 2.442410707473755, "learning_rate": 3.3333333333333333e-06, "num_tokens": 4965596.0, "completions/mean_length": 69.125, "completions/min_length": 69.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.125, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9970394968986511, "rewards/meter/std": 0.0007004055660218, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9970394968986511, "rewards/total_composite/std": 0.0007004055660218, "reward": 0.9970394968986511, "reward_std": 0.0007004134822636843, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005273114424198866, "sampling/sampling_logp_difference/max": 0.7160444259643555, "sampling/importance_sampling_ratio/min": 0.48868146538734436, "sampling/importance_sampling_ratio/mean": 1.001259207725525, "sampling/importance_sampling_ratio/max": 1.3967480659484863, "entropy": 0.029302328824996948, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9970394968986511, "reward_meter_mean": 0.9970394968986511, "reward_meter_std": 0.0007004055660218, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9970394968986511, "reward_total_composite_std": 0.0007004055660218} {"timestamp_utc": "2026-04-12T01:38:17Z", "mode": "train", "global_step": 2202, "epoch": 0.08844439089046872, "loss": 0.0069, "grad_norm": 2.4785351753234863, "learning_rate": 3.3303030303030306e-06, "num_tokens": 4969547.0, "completions/mean_length": 277.875, "completions/min_length": 273.0, "completions/max_length": 285.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 277.875, "completions/min_terminated_length": 273.0, "completions/max_terminated_length": 285.0, "rewards/meter/mean": 0.997477114200592, "rewards/meter/std": 0.001005422556772828, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9519230723381042, "rewards/repeat_penalty/std": 0.05723259970545769, "rewards/total_composite/mean": 0.8308628797531128, "rewards/total_composite/std": 0.05055442824959755, "reward": 0.8308628797531128, "reward_std": 0.05055440962314606, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.046521060168743134, "sampling/sampling_logp_difference/max": 1.3806867599487305, "sampling/importance_sampling_ratio/min": 0.25140586495399475, "sampling/importance_sampling_ratio/mean": 1.0075916051864624, "sampling/importance_sampling_ratio/max": 1.8519618511199951, "entropy": 0.4109150692820549, "clip_ratio/low_mean": 0.014749160269275308, "clip_ratio/low_min": 0.014749160269275308, "clip_ratio/high_mean": 0.020743532106280327, "clip_ratio/high_max": 0.020743532106280327, "clip_ratio/region_mean": 0.035492692375555634, "reward_total_mean": 0.8308628797531128, "reward_meter_mean": 0.997477114200592, "reward_meter_std": 0.001005422556772828, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9519230723381042, "reward_repeat_penalty_std": 0.05723259970545769, "reward_total_composite_mean": 0.8308628797531128, "reward_total_composite_std": 0.05055442824959755} {"timestamp_utc": "2026-04-12T01:38:22Z", "mode": "train", "global_step": 2203, "epoch": 0.08848455637225369, "loss": 0.0081, "grad_norm": 3.297503709793091, "learning_rate": 3.3272727272727274e-06, "num_tokens": 4971474.0, "completions/mean_length": 73.875, "completions/min_length": 71.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.875, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9979462623596191, "rewards/meter/std": 0.002537182066589594, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979462623596191, "rewards/total_composite/std": 0.002537182066589594, "reward": 0.9979462623596191, "reward_std": 0.002537175314500928, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.060300085693597794, "sampling/sampling_logp_difference/max": 1.7386038303375244, "sampling/importance_sampling_ratio/min": 0.17576563358306885, "sampling/importance_sampling_ratio/mean": 1.004874587059021, "sampling/importance_sampling_ratio/max": 1.8745375871658325, "entropy": 0.46222241409122944, "clip_ratio/low_mean": 0.0050675676902756095, "clip_ratio/low_min": 0.0050675676902756095, "clip_ratio/high_mean": 0.030337184784002602, "clip_ratio/high_max": 0.030337184784002602, "clip_ratio/region_mean": 0.03540475247427821, "reward_total_mean": 0.9979462623596191, "reward_meter_mean": 0.9979462623596191, "reward_meter_std": 0.002537182066589594, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9979462623596191, "reward_total_composite_std": 0.002537182066589594} {"timestamp_utc": "2026-04-12T01:38:31Z", "mode": "train", "global_step": 2204, "epoch": 0.08852472185403865, "loss": -0.0316, "grad_norm": 1.867836833000183, "learning_rate": 3.3242424242424242e-06, "num_tokens": 4976337.0, "completions/mean_length": 410.875, "completions/min_length": 380.0, "completions/max_length": 449.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 410.875, "completions/min_terminated_length": 380.0, "completions/max_terminated_length": 449.0, "rewards/meter/mean": 0.9976021647453308, "rewards/meter/std": 0.0008352479781024158, "rewards/count_adherence/mean": 0.6029411554336548, "rewards/count_adherence/std": 0.027230001986026764, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8183584213256836, "rewards/repeat_penalty/std": 0.10606583207845688, "rewards/total_composite/mean": 0.4941902458667755, "rewards/total_composite/std": 0.08418715745210648, "reward": 0.4941902458667755, "reward_std": 0.08418715000152588, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0467507466673851, "sampling/sampling_logp_difference/max": 6.5194783210754395, "sampling/importance_sampling_ratio/min": 0.0014744381187483668, "sampling/importance_sampling_ratio/mean": 1.0103265047073364, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3659870997071266, "clip_ratio/low_mean": 0.014909796707797796, "clip_ratio/low_min": 0.014909796707797796, "clip_ratio/high_mean": 0.014851224608719349, "clip_ratio/high_max": 0.014851224608719349, "clip_ratio/region_mean": 0.029761021316517144, "reward_total_mean": 0.4941902458667755, "reward_meter_mean": 0.9976021647453308, "reward_meter_std": 0.0008352479781024158, "reward_count_adherence_mean": 0.6029411554336548, "reward_count_adherence_std": 0.027230001986026764, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8183584213256836, "reward_repeat_penalty_std": 0.10606583207845688, "reward_total_composite_mean": 0.4941902458667755, "reward_total_composite_std": 0.08418715745210648} {"timestamp_utc": "2026-04-12T01:38:38Z", "mode": "train", "global_step": 2205, "epoch": 0.0885648873358236, "loss": 0.0016, "grad_norm": 1.170793056488037, "learning_rate": 3.321212121212121e-06, "num_tokens": 4979646.0, "completions/mean_length": 223.625, "completions/min_length": 223.0, "completions/max_length": 225.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 223.625, "completions/min_terminated_length": 223.0, "completions/max_terminated_length": 225.0, "rewards/meter/mean": 0.9960297346115112, "rewards/meter/std": 0.0002207064680987969, "rewards/count_adherence/mean": 0.8888888955116272, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5902398228645325, "rewards/total_composite/std": 0.00013078592019155622, "reward": 0.5902398228645325, "reward_std": 0.00013079482596367598, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005620577838271856, "sampling/sampling_logp_difference/max": 1.2731132507324219, "sampling/importance_sampling_ratio/min": 0.27995866537094116, "sampling/importance_sampling_ratio/mean": 1.0011334419250488, "sampling/importance_sampling_ratio/max": 1.5353397130966187, "entropy": 0.02647018409334123, "clip_ratio/low_mean": 0.0005555555690079927, "clip_ratio/low_min": 0.0005555555690079927, "clip_ratio/high_mean": 0.0022346453624777496, "clip_ratio/high_max": 0.0022346453624777496, "clip_ratio/region_mean": 0.0027902009314857423, "reward_total_mean": 0.5902398228645325, "reward_meter_mean": 0.9960297346115112, "reward_meter_std": 0.0002207064680987969, "reward_count_adherence_mean": 0.8888888955116272, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5902398228645325, "reward_total_composite_std": 0.00013078592019155622} {"timestamp_utc": "2026-04-12T01:38:43Z", "mode": "train", "global_step": 2206, "epoch": 0.08860505281760855, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.3181818181818188e-06, "num_tokens": 4981398.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0008829182479530573, "sampling/sampling_logp_difference/max": 0.027269084006547928, "sampling/importance_sampling_ratio/min": 0.9798296689987183, "sampling/importance_sampling_ratio/mean": 1.0007154941558838, "sampling/importance_sampling_ratio/max": 1.0276442766189575, "entropy": 0.009517844882793725, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:38:52Z", "mode": "train", "global_step": 2207, "epoch": 0.08864521829939351, "loss": -0.0214, "grad_norm": 1.6692476272583008, "learning_rate": 3.3151515151515156e-06, "num_tokens": 4986186.0, "completions/mean_length": 377.5, "completions/min_length": 362.0, "completions/max_length": 396.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 377.5, "completions/min_terminated_length": 362.0, "completions/max_terminated_length": 396.0, "rewards/meter/mean": 0.9973218441009521, "rewards/meter/std": 0.000687557680066675, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.0345032773911953, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8846749067306519, "rewards/repeat_penalty/std": 0.11391878128051758, "rewards/total_composite/mean": 0.5530020594596863, "rewards/total_composite/std": 0.08831970393657684, "reward": 0.5530020594596863, "reward_std": 0.08831970393657684, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03707325831055641, "sampling/sampling_logp_difference/max": 2.2571840286254883, "sampling/importance_sampling_ratio/min": 0.10464474558830261, "sampling/importance_sampling_ratio/mean": 1.0060136318206787, "sampling/importance_sampling_ratio/max": 1.8353766202926636, "entropy": 0.3299342021346092, "clip_ratio/low_mean": 0.0101922721369192, "clip_ratio/low_min": 0.0101922721369192, "clip_ratio/high_mean": 0.019536115461960435, "clip_ratio/high_max": 0.019536115461960435, "clip_ratio/region_mean": 0.029728387598879635, "reward_total_mean": 0.5530020594596863, "reward_meter_mean": 0.9973218441009521, "reward_meter_std": 0.000687557680066675, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.0345032773911953, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8846749067306519, "reward_repeat_penalty_std": 0.11391878128051758, "reward_total_composite_mean": 0.5530020594596863, "reward_total_composite_std": 0.08831970393657684} {"timestamp_utc": "2026-04-12T01:38:56Z", "mode": "train", "global_step": 2208, "epoch": 0.08868538378117846, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.3121212121212124e-06, "num_tokens": 4988010.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0008998184930533171, "sampling/sampling_logp_difference/max": 0.027839738875627518, "sampling/importance_sampling_ratio/min": 0.9878785610198975, "sampling/importance_sampling_ratio/mean": 1.0008260011672974, "sampling/importance_sampling_ratio/max": 1.028230905532837, "entropy": 0.00906775618204847, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:39:01Z", "mode": "train", "global_step": 2209, "epoch": 0.08872554926296342, "loss": -0.0012, "grad_norm": 2.7090210914611816, "learning_rate": 3.3090909090909097e-06, "num_tokens": 4989794.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.7749799489974976, "rewards/meter/std": 0.033681225031614304, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7749799489974976, "rewards/total_composite/std": 0.033681225031614304, "reward": 0.7749799489974976, "reward_std": 0.033681221306324005, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00368741643615067, "sampling/sampling_logp_difference/max": 0.8190579414367676, "sampling/importance_sampling_ratio/min": 0.4408467411994934, "sampling/importance_sampling_ratio/mean": 0.9990789890289307, "sampling/importance_sampling_ratio/max": 1.0404185056686401, "entropy": 0.011399149021599442, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7749799489974976, "reward_meter_mean": 0.7749799489974976, "reward_meter_std": 0.033681225031614304, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7749799489974976, "reward_total_composite_std": 0.033681225031614304} {"timestamp_utc": "2026-04-12T01:39:06Z", "mode": "train", "global_step": 2210, "epoch": 0.08876571474474837, "loss": 0.0121, "grad_norm": 6.713454723358154, "learning_rate": 3.3060606060606065e-06, "num_tokens": 4991661.0, "completions/mean_length": 80.375, "completions/min_length": 78.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.375, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.9961856007575989, "rewards/meter/std": 0.0025163544341921806, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9961856007575989, "rewards/total_composite/std": 0.0025163544341921806, "reward": 0.9961856007575989, "reward_std": 0.0025163493119180202, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.039697032421827316, "sampling/sampling_logp_difference/max": 1.445500373840332, "sampling/importance_sampling_ratio/min": 0.2356281578540802, "sampling/importance_sampling_ratio/mean": 1.0097302198410034, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3479005917906761, "clip_ratio/low_mean": 0.018732663709670305, "clip_ratio/low_min": 0.018732663709670305, "clip_ratio/high_mean": 0.01074601011350751, "clip_ratio/high_max": 0.01074601011350751, "clip_ratio/region_mean": 0.029478673823177814, "reward_total_mean": 0.9961856007575989, "reward_meter_mean": 0.9961856007575989, "reward_meter_std": 0.0025163544341921806, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9961856007575989, "reward_total_composite_std": 0.0025163544341921806} {"timestamp_utc": "2026-04-12T01:39:11Z", "mode": "train", "global_step": 2211, "epoch": 0.08880588022653332, "loss": 0.0013, "grad_norm": 0.7238816022872925, "learning_rate": 3.3030303030303033e-06, "num_tokens": 4993997.0, "completions/mean_length": 111.0, "completions/min_length": 111.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 111.0, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.9956170320510864, "rewards/meter/std": 6.146137457108125e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8533860445022583, "rewards/total_composite/std": 5.2672676247311756e-05, "reward": 0.8533860445022583, "reward_std": 5.2673178288387135e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007354200817644596, "sampling/sampling_logp_difference/max": 0.7475004196166992, "sampling/importance_sampling_ratio/min": 0.4735487401485443, "sampling/importance_sampling_ratio/mean": 0.9998362064361572, "sampling/importance_sampling_ratio/max": 1.6665387153625488, "entropy": 0.03893151739612222, "clip_ratio/low_mean": 0.0022522523067891598, "clip_ratio/low_min": 0.0022522523067891598, "clip_ratio/high_mean": 0.0045045046135783195, "clip_ratio/high_max": 0.0045045046135783195, "clip_ratio/region_mean": 0.006756756920367479, "reward_total_mean": 0.8533860445022583, "reward_meter_mean": 0.9956170320510864, "reward_meter_std": 6.146137457108125e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8533860445022583, "reward_total_composite_std": 5.2672676247311756e-05} {"timestamp_utc": "2026-04-12T01:39:18Z", "mode": "train", "global_step": 2212, "epoch": 0.08884604570831828, "loss": 0.0064, "grad_norm": 1.9566975831985474, "learning_rate": 3.3000000000000006e-06, "num_tokens": 4998173.0, "completions/mean_length": 276.0, "completions/min_length": 274.0, "completions/max_length": 277.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 276.0, "completions/min_terminated_length": 274.0, "completions/max_terminated_length": 277.0, "rewards/meter/mean": 0.9973146319389343, "rewards/meter/std": 0.0006697024800814688, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8666666746139526, "rewards/repeat_penalty/std": 0.06172133609652519, "rewards/total_composite/mean": 0.5762275457382202, "rewards/total_composite/std": 0.04107243940234184, "reward": 0.5762275457382202, "reward_std": 0.041072458028793335, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015320011414587498, "sampling/sampling_logp_difference/max": 1.8627681732177734, "sampling/importance_sampling_ratio/min": 0.17606841027736664, "sampling/importance_sampling_ratio/mean": 1.0007983446121216, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06808455288410187, "clip_ratio/low_mean": 0.005421726265922189, "clip_ratio/low_min": 0.005421726265922189, "clip_ratio/high_mean": 0.006823194562457502, "clip_ratio/high_max": 0.006823194562457502, "clip_ratio/region_mean": 0.01224492082837969, "reward_total_mean": 0.5762275457382202, "reward_meter_mean": 0.9973146319389343, "reward_meter_std": 0.0006697024800814688, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8666666746139526, "reward_repeat_penalty_std": 0.06172133609652519, "reward_total_composite_mean": 0.5762275457382202, "reward_total_composite_std": 0.04107243940234184} {"timestamp_utc": "2026-04-12T01:39:23Z", "mode": "train", "global_step": 2213, "epoch": 0.08888621119010323, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.2969696969696974e-06, "num_tokens": 4999621.0, "completions/mean_length": 29.0, "completions/min_length": 29.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9929623007774353, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9929623007774353, "rewards/total_composite/std": 0.0, "reward": 0.9929623007774353, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.000746682460885495, "sampling/sampling_logp_difference/max": 0.023203659802675247, "sampling/importance_sampling_ratio/min": 0.9999908804893494, "sampling/importance_sampling_ratio/mean": 1.000752329826355, "sampling/importance_sampling_ratio/max": 1.023474931716919, "entropy": 0.004421527351951227, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9929623007774353, "reward_meter_mean": 0.9929623007774353, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9929623007774353, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:39:30Z", "mode": "train", "global_step": 2214, "epoch": 0.08892637667188819, "loss": 0.0124, "grad_norm": 2.7332818508148193, "learning_rate": 3.2939393939393943e-06, "num_tokens": 5003989.0, "completions/mean_length": 306.0, "completions/min_length": 291.0, "completions/max_length": 319.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 306.0, "completions/min_terminated_length": 291.0, "completions/max_terminated_length": 319.0, "rewards/meter/mean": 0.996366024017334, "rewards/meter/std": 0.004282170441001654, "rewards/count_adherence/mean": 0.6477272510528564, "rewards/count_adherence/std": 0.03214123100042343, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9187728762626648, "rewards/repeat_penalty/std": 0.04368516802787781, "rewards/total_composite/mean": 0.5923222303390503, "rewards/total_composite/std": 0.027357451617717743, "reward": 0.5923222303390503, "reward_std": 0.027357440441846848, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04633094370365143, "sampling/sampling_logp_difference/max": 1.9842910766601562, "sampling/importance_sampling_ratio/min": 0.1374780386686325, "sampling/importance_sampling_ratio/mean": 1.0054079294204712, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3657448794692755, "clip_ratio/low_mean": 0.025091978488489985, "clip_ratio/low_min": 0.025091978488489985, "clip_ratio/high_mean": 0.015862172469496727, "clip_ratio/high_max": 0.015862172469496727, "clip_ratio/region_mean": 0.04095415095798671, "reward_total_mean": 0.5923222303390503, "reward_meter_mean": 0.996366024017334, "reward_meter_std": 0.004282170441001654, "reward_count_adherence_mean": 0.6477272510528564, "reward_count_adherence_std": 0.03214123100042343, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9187728762626648, "reward_repeat_penalty_std": 0.04368516802787781, "reward_total_composite_mean": 0.5923222303390503, "reward_total_composite_std": 0.027357451617717743} {"timestamp_utc": "2026-04-12T01:39:36Z", "mode": "train", "global_step": 2215, "epoch": 0.08896654215367314, "loss": -0.0075, "grad_norm": 3.505354404449463, "learning_rate": 3.290909090909091e-06, "num_tokens": 5006191.0, "completions/mean_length": 122.25, "completions/min_length": 119.0, "completions/max_length": 127.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.25, "completions/min_terminated_length": 119.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.9975504875183105, "rewards/meter/std": 0.0010059267515316606, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9975504875183105, "rewards/total_composite/std": 0.0010059267515316606, "reward": 0.9975504875183105, "reward_std": 0.001005922444164753, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04226119443774223, "sampling/sampling_logp_difference/max": 1.1195478439331055, "sampling/importance_sampling_ratio/min": 0.3264273405075073, "sampling/importance_sampling_ratio/mean": 1.0016958713531494, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3795941434800625, "clip_ratio/low_mean": 0.011416577035561204, "clip_ratio/low_min": 0.011416577035561204, "clip_ratio/high_mean": 0.030406045261770487, "clip_ratio/high_max": 0.030406045261770487, "clip_ratio/region_mean": 0.04182262229733169, "reward_total_mean": 0.9975504875183105, "reward_meter_mean": 0.9975504875183105, "reward_meter_std": 0.0010059267515316606, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9975504875183105, "reward_total_composite_std": 0.0010059267515316606} {"timestamp_utc": "2026-04-12T01:39:40Z", "mode": "train", "global_step": 2216, "epoch": 0.0890067076354581, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.2878787878787883e-06, "num_tokens": 5007999.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9992372989654541, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992372989654541, "rewards/total_composite/std": 0.0, "reward": 0.9992372989654541, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0007629028987139463, "sampling/sampling_logp_difference/max": 0.029877394437789917, "sampling/importance_sampling_ratio/min": 0.9758114814758301, "sampling/importance_sampling_ratio/mean": 1.0005970001220703, "sampling/importance_sampling_ratio/max": 1.0303282737731934, "entropy": 0.008104118285700679, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9992372989654541, "reward_meter_mean": 0.9992372989654541, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992372989654541, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:39:45Z", "mode": "train", "global_step": 2217, "epoch": 0.08904687311724305, "loss": -0.0229, "grad_norm": 3.7353017330169678, "learning_rate": 3.284848484848485e-06, "num_tokens": 5010225.0, "completions/mean_length": 108.25, "completions/min_length": 102.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.25, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.9519559144973755, "rewards/meter/std": 0.09988751262426376, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.8596517443656921, "rewards/total_composite/std": 0.15026217699050903, "reward": 0.8596517443656921, "reward_std": 0.15026216208934784, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02141641452908516, "sampling/sampling_logp_difference/max": 1.004293441772461, "sampling/importance_sampling_ratio/min": 0.36630338430404663, "sampling/importance_sampling_ratio/mean": 1.003779649734497, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.14850023202598095, "clip_ratio/low_mean": 0.005999947316013277, "clip_ratio/low_min": 0.005999947316013277, "clip_ratio/high_mean": 0.014177569886669517, "clip_ratio/high_max": 0.014177569886669517, "clip_ratio/region_mean": 0.020177517202682793, "reward_total_mean": 0.8596517443656921, "reward_meter_mean": 0.9519559144973755, "reward_meter_std": 0.09988751262426376, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_total_composite_mean": 0.8596517443656921, "reward_total_composite_std": 0.15026217699050903} {"timestamp_utc": "2026-04-12T01:39:55Z", "mode": "train", "global_step": 2218, "epoch": 0.089087038599028, "loss": -0.003, "grad_norm": 1.9956766366958618, "learning_rate": 3.281818181818182e-06, "num_tokens": 5015622.0, "completions/mean_length": 474.625, "completions/min_length": 460.0, "completions/max_length": 487.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 474.625, "completions/min_terminated_length": 460.0, "completions/max_terminated_length": 487.0, "rewards/meter/mean": 0.9962097406387329, "rewards/meter/std": 0.003824215615168214, "rewards/count_adherence/mean": 0.5986841917037964, "rewards/count_adherence/std": 0.027239417657256126, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8791407942771912, "rewards/repeat_penalty/std": 0.11821687966585159, "rewards/total_composite/mean": 0.5244148969650269, "rewards/total_composite/std": 0.0736735388636589, "reward": 0.5244148969650269, "reward_std": 0.07367353141307831, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04105058312416077, "sampling/sampling_logp_difference/max": 1.8464699983596802, "sampling/importance_sampling_ratio/min": 0.15779319405555725, "sampling/importance_sampling_ratio/mean": 1.007934808731079, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.28974736854434013, "clip_ratio/low_mean": 0.011183922295458615, "clip_ratio/low_min": 0.011183922295458615, "clip_ratio/high_mean": 0.017237887950614095, "clip_ratio/high_max": 0.017237887950614095, "clip_ratio/region_mean": 0.02842181024607271, "reward_total_mean": 0.5244148969650269, "reward_meter_mean": 0.9962097406387329, "reward_meter_std": 0.003824215615168214, "reward_count_adherence_mean": 0.5986841917037964, "reward_count_adherence_std": 0.027239417657256126, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8791407942771912, "reward_repeat_penalty_std": 0.11821687966585159, "reward_total_composite_mean": 0.5244148969650269, "reward_total_composite_std": 0.0736735388636589} {"timestamp_utc": "2026-04-12T01:40:00Z", "mode": "train", "global_step": 2219, "epoch": 0.08912720408081296, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.2787878787878793e-06, "num_tokens": 5017335.0, "completions/mean_length": 64.125, "completions/min_length": 64.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.125, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9992372989654541, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992372989654541, "rewards/total_composite/std": 0.0, "reward": 0.9992372989654541, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0031146432738751173, "sampling/sampling_logp_difference/max": 0.3054380416870117, "sampling/importance_sampling_ratio/min": 0.7368006110191345, "sampling/importance_sampling_ratio/mean": 1.000244140625, "sampling/importance_sampling_ratio/max": 1.1355178356170654, "entropy": 0.01567571924533695, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9992372989654541, "reward_meter_mean": 0.9992372989654541, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992372989654541, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:40:05Z", "mode": "train", "global_step": 2220, "epoch": 0.08916736956259791, "loss": 0.0018, "grad_norm": 2.9601826667785645, "learning_rate": 3.275757575757576e-06, "num_tokens": 5019823.0, "completions/mean_length": 137.0, "completions/min_length": 137.0, "completions/max_length": 137.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 137.0, "completions/min_terminated_length": 137.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.9971017837524414, "rewards/meter/std": 0.0003135971783194691, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8546586632728577, "rewards/total_composite/std": 0.0002687963715288788, "reward": 0.8546586632728577, "reward_std": 0.00026877920026890934, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003638867987319827, "sampling/sampling_logp_difference/max": 0.5708838701248169, "sampling/importance_sampling_ratio/min": 0.565025806427002, "sampling/importance_sampling_ratio/mean": 0.9999095797538757, "sampling/importance_sampling_ratio/max": 1.3408164978027344, "entropy": 0.018264403683133423, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8546586632728577, "reward_meter_mean": 0.9971017837524414, "reward_meter_std": 0.0003135971783194691, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8546586632728577, "reward_total_composite_std": 0.0002687963715288788} {"timestamp_utc": "2026-04-12T01:40:10Z", "mode": "train", "global_step": 2221, "epoch": 0.08920753504438286, "loss": -0.0001, "grad_norm": 0.8054800033569336, "learning_rate": 3.272727272727273e-06, "num_tokens": 5021798.0, "completions/mean_length": 82.875, "completions/min_length": 82.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.875, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.9953393936157227, "rewards/meter/std": 4.823669223696925e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9953393936157227, "rewards/total_composite/std": 4.823669223696925e-05, "reward": 0.9953393936157227, "reward_std": 4.822280607186258e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009308185428380966, "sampling/sampling_logp_difference/max": 0.9146785736083984, "sampling/importance_sampling_ratio/min": 0.4006454050540924, "sampling/importance_sampling_ratio/mean": 0.99955153465271, "sampling/importance_sampling_ratio/max": 1.2980178594589233, "entropy": 0.03914514696225524, "clip_ratio/low_mean": 0.0015243901871144772, "clip_ratio/low_min": 0.0015243901871144772, "clip_ratio/high_mean": 0.0015060240402817726, "clip_ratio/high_max": 0.0015060240402817726, "clip_ratio/region_mean": 0.0030304142273962498, "reward_total_mean": 0.9953393936157227, "reward_meter_mean": 0.9953393936157227, "reward_meter_std": 4.823669223696925e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9953393936157227, "reward_total_composite_std": 4.823669223696925e-05} {"timestamp_utc": "2026-04-12T01:40:17Z", "mode": "train", "global_step": 2222, "epoch": 0.08924770052616782, "loss": 0.0058, "grad_norm": 1.5890640020370483, "learning_rate": 3.2696969696969698e-06, "num_tokens": 5025785.0, "completions/mean_length": 290.375, "completions/min_length": 287.0, "completions/max_length": 293.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 290.375, "completions/min_terminated_length": 287.0, "completions/max_terminated_length": 293.0, "rewards/meter/mean": 0.9988961219787598, "rewards/meter/std": 0.0002896705409511924, "rewards/count_adherence/mean": 0.7272727489471436, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9166666269302368, "rewards/repeat_penalty/std": 0.0471404492855072, "rewards/total_composite/mean": 0.6659300327301025, "rewards/total_composite/std": 0.0342310331761837, "reward": 0.6659300327301025, "reward_std": 0.03423101827502251, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02183903567492962, "sampling/sampling_logp_difference/max": 1.4761362075805664, "sampling/importance_sampling_ratio/min": 0.22851894795894623, "sampling/importance_sampling_ratio/mean": 1.0035452842712402, "sampling/importance_sampling_ratio/max": 1.77455735206604, "entropy": 0.24721034429967403, "clip_ratio/low_mean": 0.006000505993142724, "clip_ratio/low_min": 0.006000505993142724, "clip_ratio/high_mean": 0.01464845088776201, "clip_ratio/high_max": 0.01464845088776201, "clip_ratio/region_mean": 0.020648956880904734, "reward_total_mean": 0.6659300327301025, "reward_meter_mean": 0.9988961219787598, "reward_meter_std": 0.0002896705409511924, "reward_count_adherence_mean": 0.7272727489471436, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9166666269302368, "reward_repeat_penalty_std": 0.0471404492855072, "reward_total_composite_mean": 0.6659300327301025, "reward_total_composite_std": 0.0342310331761837} {"timestamp_utc": "2026-04-12T01:40:23Z", "mode": "train", "global_step": 2223, "epoch": 0.08928786600795277, "loss": 0.0074, "grad_norm": 1.8738999366760254, "learning_rate": 3.266666666666667e-06, "num_tokens": 5028644.0, "completions/mean_length": 172.375, "completions/min_length": 171.0, "completions/max_length": 182.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 172.375, "completions/min_terminated_length": 171.0, "completions/max_terminated_length": 182.0, "rewards/meter/mean": 0.9511035084724426, "rewards/meter/std": 0.01162177324295044, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7777777910232544, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6164560317993164, "rewards/total_composite/std": 0.00753263384103775, "reward": 0.6164560317993164, "reward_std": 0.0075326296500861645, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012546628713607788, "sampling/sampling_logp_difference/max": 0.6718964576721191, "sampling/importance_sampling_ratio/min": 0.5107390880584717, "sampling/importance_sampling_ratio/mean": 1.002518892288208, "sampling/importance_sampling_ratio/max": 1.608762502670288, "entropy": 0.0778789222240448, "clip_ratio/low_mean": 0.005715410457924008, "clip_ratio/low_min": 0.005715410457924008, "clip_ratio/high_mean": 0.002923976629972458, "clip_ratio/high_max": 0.002923976629972458, "clip_ratio/region_mean": 0.008639387087896466, "reward_total_mean": 0.6164560317993164, "reward_meter_mean": 0.9511035084724426, "reward_meter_std": 0.01162177324295044, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7777777910232544, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6164560317993164, "reward_total_composite_std": 0.00753263384103775} {"timestamp_utc": "2026-04-12T01:40:28Z", "mode": "train", "global_step": 2224, "epoch": 0.08932803148973772, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.263636363636364e-06, "num_tokens": 5030428.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0009967123623937368, "sampling/sampling_logp_difference/max": 0.025670286267995834, "sampling/importance_sampling_ratio/min": 0.9967485070228577, "sampling/importance_sampling_ratio/mean": 1.000982403755188, "sampling/importance_sampling_ratio/max": 1.0260026454925537, "entropy": 0.00872120977146551, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:40:32Z", "mode": "train", "global_step": 2225, "epoch": 0.08936819697152268, "loss": 0.0108, "grad_norm": 3.8838281631469727, "learning_rate": 3.2606060606060607e-06, "num_tokens": 5032200.0, "completions/mean_length": 76.5, "completions/min_length": 73.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.5, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9987069964408875, "rewards/meter/std": 0.0007462438079528511, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9987069964408875, "rewards/total_composite/std": 0.0007462438079528511, "reward": 0.9987069964408875, "reward_std": 0.0007462415378540754, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03162068501114845, "sampling/sampling_logp_difference/max": 0.9698367118835449, "sampling/importance_sampling_ratio/min": 0.37914493680000305, "sampling/importance_sampling_ratio/mean": 1.0066579580307007, "sampling/importance_sampling_ratio/max": 1.6162910461425781, "entropy": 0.27160678431391716, "clip_ratio/low_mean": 0.0032748287776485085, "clip_ratio/low_min": 0.0032748287776485085, "clip_ratio/high_mean": 0.021526333526708186, "clip_ratio/high_max": 0.021526333526708186, "clip_ratio/region_mean": 0.024801162304356694, "reward_total_mean": 0.9987069964408875, "reward_meter_mean": 0.9987069964408875, "reward_meter_std": 0.0007462438079528511, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9987069964408875, "reward_total_composite_std": 0.0007462438079528511} {"timestamp_utc": "2026-04-12T01:40:37Z", "mode": "train", "global_step": 2226, "epoch": 0.08940836245330763, "loss": 0.0153, "grad_norm": 7.719726085662842, "learning_rate": 3.257575757575758e-06, "num_tokens": 5033803.0, "completions/mean_length": 36.375, "completions/min_length": 36.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.375, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9948956370353699, "rewards/meter/std": 0.005730676930397749, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9948956370353699, "rewards/total_composite/std": 0.005730676930397749, "reward": 0.9948956370353699, "reward_std": 0.005730684380978346, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028130551800131798, "sampling/sampling_logp_difference/max": 0.9965043067932129, "sampling/importance_sampling_ratio/min": 0.369167685508728, "sampling/importance_sampling_ratio/mean": 1.006754994392395, "sampling/importance_sampling_ratio/max": 1.3892041444778442, "entropy": 0.18548811599612236, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.010416666744276881, "clip_ratio/high_max": 0.010416666744276881, "clip_ratio/region_mean": 0.010416666744276881, "reward_total_mean": 0.9948956370353699, "reward_meter_mean": 0.9948956370353699, "reward_meter_std": 0.005730676930397749, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9948956370353699, "reward_total_composite_std": 0.005730676930397749} {"timestamp_utc": "2026-04-12T01:40:47Z", "mode": "train", "global_step": 2227, "epoch": 0.08944852793509259, "loss": 0.205, "grad_norm": 1.227808952331543, "learning_rate": 3.2545454545454548e-06, "num_tokens": 5038942.0, "completions/mean_length": 477.375, "completions/min_length": 445.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 472.4285888671875, "completions/min_terminated_length": 445.0, "completions/max_terminated_length": 505.0, "rewards/meter/mean": 0.9960645437240601, "rewards/meter/std": 0.0029486457351595163, "rewards/count_adherence/mean": 0.5625, "rewards/count_adherence/std": 0.0353553481400013, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8150820732116699, "rewards/repeat_penalty/std": 0.08796744048595428, "rewards/total_composite/mean": 0.4574677348136902, "rewards/total_composite/std": 0.06483428180217743, "reward": 0.4574677348136902, "reward_std": 0.06483428180217743, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03101653791964054, "sampling/sampling_logp_difference/max": 2.6449055671691895, "sampling/importance_sampling_ratio/min": 0.07101205736398697, "sampling/importance_sampling_ratio/mean": 1.0072216987609863, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.21459073200821877, "clip_ratio/low_mean": 0.014457646873779595, "clip_ratio/low_min": 0.014457646873779595, "clip_ratio/high_mean": 0.005880667129531503, "clip_ratio/high_max": 0.005880667129531503, "clip_ratio/region_mean": 0.020338314003311098, "reward_total_mean": 0.4574677348136902, "reward_meter_mean": 0.9960645437240601, "reward_meter_std": 0.0029486457351595163, "reward_count_adherence_mean": 0.5625, "reward_count_adherence_std": 0.0353553481400013, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8150820732116699, "reward_repeat_penalty_std": 0.08796744048595428, "reward_total_composite_mean": 0.4574677348136902, "reward_total_composite_std": 0.06483428180217743} {"timestamp_utc": "2026-04-12T01:40:51Z", "mode": "train", "global_step": 2228, "epoch": 0.08948869341687754, "loss": 0.0001, "grad_norm": 0.01318140048533678, "learning_rate": 3.2515151515151516e-06, "num_tokens": 5040574.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.7876332998275757, "rewards/meter/std": 1.5573279597447254e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7876332998275757, "rewards/total_composite/std": 1.5573279597447254e-05, "reward": 0.7876332998275757, "reward_std": 1.558229632792063e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0009326103026978672, "sampling/sampling_logp_difference/max": 0.06658339500427246, "sampling/importance_sampling_ratio/min": 0.991096556186676, "sampling/importance_sampling_ratio/mean": 1.0008612871170044, "sampling/importance_sampling_ratio/max": 1.068850040435791, "entropy": 0.007959658920299262, "clip_ratio/low_mean": 0.002314814832061529, "clip_ratio/low_min": 0.002314814832061529, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.002314814832061529, "reward_total_mean": 0.7876332998275757, "reward_meter_mean": 0.7876332998275757, "reward_meter_std": 1.5573279597447254e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7876332998275757, "reward_total_composite_std": 1.5573279597447254e-05} {"timestamp_utc": "2026-04-12T01:40:57Z", "mode": "train", "global_step": 2229, "epoch": 0.0895288588986625, "loss": -0.0005, "grad_norm": 2.3057451248168945, "learning_rate": 3.2484848484848484e-06, "num_tokens": 5043518.0, "completions/mean_length": 166.0, "completions/min_length": 165.0, "completions/max_length": 167.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 166.0, "completions/min_terminated_length": 165.0, "completions/max_terminated_length": 167.0, "rewards/meter/mean": 0.9984830617904663, "rewards/meter/std": 6.16059624007903e-05, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8611111640930176, "rewards/repeat_penalty/std": 0.05143444612622261, "rewards/total_composite/mean": 0.7165045738220215, "rewards/total_composite/std": 0.042806532233953476, "reward": 0.7165045738220215, "reward_std": 0.042806532233953476, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017393097281455994, "sampling/sampling_logp_difference/max": 2.966561794281006, "sampling/importance_sampling_ratio/min": 0.0514800027012825, "sampling/importance_sampling_ratio/mean": 1.0007810592651367, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06033759843558073, "clip_ratio/low_mean": 0.0015151514671742916, "clip_ratio/low_min": 0.0015151514671742916, "clip_ratio/high_mean": 0.010519623290747404, "clip_ratio/high_max": 0.010519623290747404, "clip_ratio/region_mean": 0.012034774757921696, "reward_total_mean": 0.7165045738220215, "reward_meter_mean": 0.9984830617904663, "reward_meter_std": 6.16059624007903e-05, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8611111640930176, "reward_repeat_penalty_std": 0.05143444612622261, "reward_total_composite_mean": 0.7165045738220215, "reward_total_composite_std": 0.042806532233953476} {"timestamp_utc": "2026-04-12T01:41:01Z", "mode": "train", "global_step": 2230, "epoch": 0.08956902438044745, "loss": 0.0099, "grad_norm": 10.35647964477539, "learning_rate": 3.2454545454545457e-06, "num_tokens": 5044997.0, "completions/mean_length": 33.875, "completions/min_length": 33.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.875, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9669080972671509, "rewards/meter/std": 0.007700720801949501, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9669080972671509, "rewards/total_composite/std": 0.007700720801949501, "reward": 0.9669080972671509, "reward_std": 0.007700727321207523, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.042054299265146255, "sampling/sampling_logp_difference/max": 2.527313470840454, "sampling/importance_sampling_ratio/min": 0.07987331598997116, "sampling/importance_sampling_ratio/mean": 1.003213882446289, "sampling/importance_sampling_ratio/max": 1.6818870306015015, "entropy": 0.1314310822635889, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.02239304850809276, "clip_ratio/high_max": 0.02239304850809276, "clip_ratio/region_mean": 0.02239304850809276, "reward_total_mean": 0.9669080972671509, "reward_meter_mean": 0.9669080972671509, "reward_meter_std": 0.007700720801949501, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9669080972671509, "reward_total_composite_std": 0.007700720801949501} {"timestamp_utc": "2026-04-12T01:41:06Z", "mode": "train", "global_step": 2231, "epoch": 0.0896091898622324, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.2424242424242425e-06, "num_tokens": 5046893.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0011728927493095398, "sampling/sampling_logp_difference/max": 0.02333247661590576, "sampling/importance_sampling_ratio/min": 0.9904347658157349, "sampling/importance_sampling_ratio/mean": 1.0010844469070435, "sampling/importance_sampling_ratio/max": 1.023606777191162, "entropy": 0.010291360202245414, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:41:10Z", "mode": "train", "global_step": 2232, "epoch": 0.08964935534401736, "loss": -0.0002, "grad_norm": 0.10460558533668518, "learning_rate": 3.2393939393939393e-06, "num_tokens": 5048693.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9992489814758301, "rewards/meter/std": 1.6926040188991465e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992489814758301, "rewards/total_composite/std": 1.6926040188991465e-05, "reward": 0.9992489814758301, "reward_std": 1.6926040188991465e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006007255986332893, "sampling/sampling_logp_difference/max": 1.3880231380462646, "sampling/importance_sampling_ratio/min": 0.24956819415092468, "sampling/importance_sampling_ratio/mean": 0.9996911883354187, "sampling/importance_sampling_ratio/max": 1.2030258178710938, "entropy": 0.019377628806978464, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0078125, "clip_ratio/high_max": 0.0078125, "clip_ratio/region_mean": 0.0078125, "reward_total_mean": 0.9992489814758301, "reward_meter_mean": 0.9992489814758301, "reward_meter_std": 1.6926040188991465e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992489814758301, "reward_total_composite_std": 1.6926040188991465e-05} {"timestamp_utc": "2026-04-12T01:41:16Z", "mode": "train", "global_step": 2233, "epoch": 0.08968952082580231, "loss": 0.0026, "grad_norm": 1.9100828170776367, "learning_rate": 3.236363636363636e-06, "num_tokens": 5051460.0, "completions/mean_length": 145.875, "completions/min_length": 143.0, "completions/max_length": 149.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 145.875, "completions/min_terminated_length": 143.0, "completions/max_terminated_length": 149.0, "rewards/meter/mean": 0.9988158941268921, "rewards/meter/std": 0.0003107816446572542, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9107142686843872, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.7277184724807739, "rewards/total_composite/std": 0.059221845120191574, "reward": 0.7277184724807739, "reward_std": 0.059221841394901276, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027437884360551834, "sampling/sampling_logp_difference/max": 1.669896125793457, "sampling/importance_sampling_ratio/min": 0.18826662003993988, "sampling/importance_sampling_ratio/mean": 1.0000747442245483, "sampling/importance_sampling_ratio/max": 1.9528251886367798, "entropy": 0.2233992125838995, "clip_ratio/low_mean": 0.012797378934919834, "clip_ratio/low_min": 0.012797378934919834, "clip_ratio/high_mean": 0.005999534041620791, "clip_ratio/high_max": 0.005999534041620791, "clip_ratio/region_mean": 0.018796912976540625, "reward_total_mean": 0.7277184724807739, "reward_meter_mean": 0.9988158941268921, "reward_meter_std": 0.0003107816446572542, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9107142686843872, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.7277184724807739, "reward_total_composite_std": 0.059221845120191574} {"timestamp_utc": "2026-04-12T01:41:21Z", "mode": "train", "global_step": 2234, "epoch": 0.08972968630758726, "loss": 0.0004, "grad_norm": 3.689906120300293, "learning_rate": 3.2333333333333334e-06, "num_tokens": 5053322.0, "completions/mean_length": 68.75, "completions/min_length": 68.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.75, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9638321399688721, "rewards/meter/std": 0.004275594372302294, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9638321399688721, "rewards/total_composite/std": 0.004275594372302294, "reward": 0.9638321399688721, "reward_std": 0.004275602288544178, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02401396632194519, "sampling/sampling_logp_difference/max": 1.2271108627319336, "sampling/importance_sampling_ratio/min": 0.2931382656097412, "sampling/importance_sampling_ratio/mean": 0.9958019852638245, "sampling/importance_sampling_ratio/max": 1.3294564485549927, "entropy": 0.10225279349833727, "clip_ratio/low_mean": 0.005488064838573337, "clip_ratio/low_min": 0.005488064838573337, "clip_ratio/high_mean": 0.017986543010920286, "clip_ratio/high_max": 0.017986543010920286, "clip_ratio/region_mean": 0.023474607849493623, "reward_total_mean": 0.9638321399688721, "reward_meter_mean": 0.9638321399688721, "reward_meter_std": 0.004275594372302294, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9638321399688721, "reward_total_composite_std": 0.004275594372302294} {"timestamp_utc": "2026-04-12T01:41:26Z", "mode": "train", "global_step": 2235, "epoch": 0.08976985178937222, "loss": 0.0039, "grad_norm": 1.4711960554122925, "learning_rate": 3.2303030303030307e-06, "num_tokens": 5055211.0, "completions/mean_length": 69.125, "completions/min_length": 69.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.125, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9970292448997498, "rewards/meter/std": 0.0009962875628843904, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9970292448997498, "rewards/total_composite/std": 0.0009962875628843904, "reward": 0.9970292448997498, "reward_std": 0.0009962901240214705, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008085141889750957, "sampling/sampling_logp_difference/max": 0.4244270324707031, "sampling/importance_sampling_ratio/min": 0.6541445255279541, "sampling/importance_sampling_ratio/mean": 1.0035878419876099, "sampling/importance_sampling_ratio/max": 1.4033783674240112, "entropy": 0.06750913895666599, "clip_ratio/low_mean": 0.0017857142956927419, "clip_ratio/low_min": 0.0017857142956927419, "clip_ratio/high_mean": 0.005434782709926367, "clip_ratio/high_max": 0.005434782709926367, "clip_ratio/region_mean": 0.007220497005619109, "reward_total_mean": 0.9970292448997498, "reward_meter_mean": 0.9970292448997498, "reward_meter_std": 0.0009962875628843904, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9970292448997498, "reward_total_composite_std": 0.0009962875628843904} {"timestamp_utc": "2026-04-12T01:41:31Z", "mode": "train", "global_step": 2236, "epoch": 0.08981001727115717, "loss": 0.0014, "grad_norm": 0.5810080766677856, "learning_rate": 3.227272727272728e-06, "num_tokens": 5057433.0, "completions/mean_length": 102.75, "completions/min_length": 102.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.75, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.9973079562187195, "rewards/meter/std": 4.1739614971447736e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973079562187195, "rewards/total_composite/std": 4.1739614971447736e-05, "reward": 0.9973079562187195, "reward_std": 4.174207424512133e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008572143502533436, "sampling/sampling_logp_difference/max": 0.9729032516479492, "sampling/importance_sampling_ratio/min": 0.39697858691215515, "sampling/importance_sampling_ratio/mean": 1.0025743246078491, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.05867503536865115, "clip_ratio/low_mean": 0.0036407767329365015, "clip_ratio/low_min": 0.0036407767329365015, "clip_ratio/high_mean": 0.006067961221560836, "clip_ratio/high_max": 0.006067961221560836, "clip_ratio/region_mean": 0.009708737954497337, "reward_total_mean": 0.9973079562187195, "reward_meter_mean": 0.9973079562187195, "reward_meter_std": 4.1739614971447736e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9973079562187195, "reward_total_composite_std": 4.1739614971447736e-05} {"timestamp_utc": "2026-04-12T01:41:36Z", "mode": "train", "global_step": 2237, "epoch": 0.08985018275294213, "loss": -0.0042, "grad_norm": 2.2996366024017334, "learning_rate": 3.2242424242424248e-06, "num_tokens": 5060180.0, "completions/mean_length": 158.375, "completions/min_length": 155.0, "completions/max_length": 164.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 158.375, "completions/min_terminated_length": 155.0, "completions/max_terminated_length": 164.0, "rewards/meter/mean": 0.9958896636962891, "rewards/meter/std": 0.0022456336300820112, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.7824714183807373, "rewards/total_composite/std": 0.040016695857048035, "reward": 0.7824714183807373, "reward_std": 0.04001671448349953, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0335751473903656, "sampling/sampling_logp_difference/max": 1.1518535614013672, "sampling/importance_sampling_ratio/min": 0.3160504102706909, "sampling/importance_sampling_ratio/mean": 1.003827452659607, "sampling/importance_sampling_ratio/max": 1.8041925430297852, "entropy": 0.28765890188515186, "clip_ratio/low_mean": 0.004807692486792803, "clip_ratio/low_min": 0.004807692486792803, "clip_ratio/high_mean": 0.021151655237190425, "clip_ratio/high_max": 0.021151655237190425, "clip_ratio/region_mean": 0.025959347723983228, "reward_total_mean": 0.7824714183807373, "reward_meter_mean": 0.9958896636962891, "reward_meter_std": 0.0022456336300820112, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.7824714183807373, "reward_total_composite_std": 0.040016695857048035} {"timestamp_utc": "2026-04-12T01:41:41Z", "mode": "train", "global_step": 2238, "epoch": 0.08989034823472708, "loss": 0.0, "grad_norm": 0.5047301650047302, "learning_rate": 3.2212121212121216e-06, "num_tokens": 5061987.0, "completions/mean_length": 68.875, "completions/min_length": 68.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.875, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9973945617675781, "rewards/meter/std": 5.974572559352964e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973945617675781, "rewards/total_composite/std": 5.974572559352964e-05, "reward": 0.9973945617675781, "reward_std": 5.973832230665721e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008767606690526009, "sampling/sampling_logp_difference/max": 0.3773939609527588, "sampling/importance_sampling_ratio/min": 0.7254200577735901, "sampling/importance_sampling_ratio/mean": 1.0013201236724854, "sampling/importance_sampling_ratio/max": 1.4584788084030151, "entropy": 0.07339224359020591, "clip_ratio/low_mean": 0.005461423774249852, "clip_ratio/low_min": 0.005461423774249852, "clip_ratio/high_mean": 0.010869565419852734, "clip_ratio/high_max": 0.010869565419852734, "clip_ratio/region_mean": 0.016330989194102585, "reward_total_mean": 0.9973945617675781, "reward_meter_mean": 0.9973945617675781, "reward_meter_std": 5.974572559352964e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9973945617675781, "reward_total_composite_std": 5.974572559352964e-05} {"timestamp_utc": "2026-04-12T01:41:46Z", "mode": "train", "global_step": 2239, "epoch": 0.08993051371651203, "loss": 0.0221, "grad_norm": 4.3548431396484375, "learning_rate": 3.2181818181818184e-06, "num_tokens": 5064107.0, "completions/mean_length": 106.0, "completions/min_length": 100.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.0, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.9938052296638489, "rewards/meter/std": 0.002556393388658762, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.8447941541671753, "rewards/total_composite/std": 0.09272917360067368, "reward": 0.8447941541671753, "reward_std": 0.09272917360067368, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02209414355456829, "sampling/sampling_logp_difference/max": 1.215738296508789, "sampling/importance_sampling_ratio/min": 0.3138589859008789, "sampling/importance_sampling_ratio/mean": 1.0017590522766113, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1653594933450222, "clip_ratio/low_mean": 0.010472985217347741, "clip_ratio/low_min": 0.010472985217347741, "clip_ratio/high_mean": 0.00717289699241519, "clip_ratio/high_max": 0.00717289699241519, "clip_ratio/region_mean": 0.01764588220976293, "reward_total_mean": 0.8447941541671753, "reward_meter_mean": 0.9938052296638489, "reward_meter_std": 0.002556393388658762, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.8447941541671753, "reward_total_composite_std": 0.09272917360067368} {"timestamp_utc": "2026-04-12T01:41:52Z", "mode": "train", "global_step": 2240, "epoch": 0.08997067919829699, "loss": 0.0227, "grad_norm": 6.652218341827393, "learning_rate": 3.2151515151515157e-06, "num_tokens": 5066446.0, "completions/mean_length": 112.375, "completions/min_length": 108.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.375, "completions/min_terminated_length": 108.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.8900635242462158, "rewards/meter/std": 0.201475590467453, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8900635242462158, "rewards/total_composite/std": 0.201475590467453, "reward": 0.8900635242462158, "reward_std": 0.2014755755662918, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05001511052250862, "sampling/sampling_logp_difference/max": 1.6565048694610596, "sampling/importance_sampling_ratio/min": 0.19080470502376556, "sampling/importance_sampling_ratio/mean": 1.0028852224349976, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3458435367792845, "clip_ratio/low_mean": 0.008410736452788115, "clip_ratio/low_min": 0.008410736452788115, "clip_ratio/high_mean": 0.036260453751310706, "clip_ratio/high_max": 0.036260453751310706, "clip_ratio/region_mean": 0.04467119020409882, "reward_total_mean": 0.8900635242462158, "reward_meter_mean": 0.8900635242462158, "reward_meter_std": 0.201475590467453, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8900635242462158, "reward_total_composite_std": 0.201475590467453} {"timestamp_utc": "2026-04-12T01:41:57Z", "mode": "train", "global_step": 2241, "epoch": 0.09001084468008194, "loss": -0.0017, "grad_norm": 2.789885997772217, "learning_rate": 3.2121212121212125e-06, "num_tokens": 5068415.0, "completions/mean_length": 74.125, "completions/min_length": 73.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.125, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9991567134857178, "rewards/meter/std": 0.0002589829673524946, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991567134857178, "rewards/total_composite/std": 0.0002589829673524946, "reward": 0.9991567134857178, "reward_std": 0.00025898346211761236, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.029841933399438858, "sampling/sampling_logp_difference/max": 1.069981575012207, "sampling/importance_sampling_ratio/min": 0.34301483631134033, "sampling/importance_sampling_ratio/mean": 1.0056432485580444, "sampling/importance_sampling_ratio/max": 1.596889853477478, "entropy": 0.2299621980637312, "clip_ratio/low_mean": 0.006711711874231696, "clip_ratio/low_min": 0.006711711874231696, "clip_ratio/high_mean": 0.01353727001696825, "clip_ratio/high_max": 0.01353727001696825, "clip_ratio/region_mean": 0.020248981891199946, "reward_total_mean": 0.9991567134857178, "reward_meter_mean": 0.9991567134857178, "reward_meter_std": 0.0002589829673524946, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9991567134857178, "reward_total_composite_std": 0.0002589829673524946} {"timestamp_utc": "2026-04-12T01:42:01Z", "mode": "train", "global_step": 2242, "epoch": 0.0900510101618669, "loss": -0.0007, "grad_norm": 1.4458215236663818, "learning_rate": 3.2090909090909094e-06, "num_tokens": 5070277.0, "completions/mean_length": 68.75, "completions/min_length": 68.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.75, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9974172115325928, "rewards/meter/std": 9.149048128165305e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9974172115325928, "rewards/total_composite/std": 9.149048128165305e-05, "reward": 0.9974172115325928, "reward_std": 9.149555262411013e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01175268180668354, "sampling/sampling_logp_difference/max": 0.9791126251220703, "sampling/importance_sampling_ratio/min": 0.3756442964076996, "sampling/importance_sampling_ratio/mean": 1.0000346899032593, "sampling/importance_sampling_ratio/max": 1.4807547330856323, "entropy": 0.06746953492984176, "clip_ratio/low_mean": 0.007326300139538944, "clip_ratio/low_min": 0.007326300139538944, "clip_ratio/high_mean": 0.007246376946568489, "clip_ratio/high_max": 0.007246376946568489, "clip_ratio/region_mean": 0.014572677086107433, "reward_total_mean": 0.9974172115325928, "reward_meter_mean": 0.9974172115325928, "reward_meter_std": 9.149048128165305e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9974172115325928, "reward_total_composite_std": 9.149048128165305e-05} {"timestamp_utc": "2026-04-12T01:42:06Z", "mode": "train", "global_step": 2243, "epoch": 0.09009117564365185, "loss": -0.0004, "grad_norm": 0.028668954968452454, "learning_rate": 3.2060606060606066e-06, "num_tokens": 5071893.0, "completions/mean_length": 55.0, "completions/min_length": 55.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.0, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9948811531066895, "rewards/meter/std": 1.4479670653599896e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9948811531066895, "rewards/total_composite/std": 1.4479670653599896e-06, "reward": 0.9948811531066895, "reward_std": 1.448755483579589e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005395011510699987, "sampling/sampling_logp_difference/max": 0.7887990474700928, "sampling/importance_sampling_ratio/min": 0.45439019799232483, "sampling/importance_sampling_ratio/mean": 0.9990595579147339, "sampling/importance_sampling_ratio/max": 1.0825896263122559, "entropy": 0.01730545365717262, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9948811531066895, "reward_meter_mean": 0.9948811531066895, "reward_meter_std": 1.4479670653599896e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9948811531066895, "reward_total_composite_std": 1.4479670653599896e-06} {"timestamp_utc": "2026-04-12T01:42:11Z", "mode": "train", "global_step": 2244, "epoch": 0.0901313411254368, "loss": 0.0324, "grad_norm": 2.5679850578308105, "learning_rate": 3.2030303030303034e-06, "num_tokens": 5073801.0, "completions/mean_length": 70.5, "completions/min_length": 68.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.5, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9620155096054077, "rewards/meter/std": 0.010956778191030025, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9620155096054077, "rewards/total_composite/std": 0.010956778191030025, "reward": 0.9620155096054077, "reward_std": 0.010956759564578533, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026699479669332504, "sampling/sampling_logp_difference/max": 1.6187305450439453, "sampling/importance_sampling_ratio/min": 0.19815008342266083, "sampling/importance_sampling_ratio/mean": 1.0068614482879639, "sampling/importance_sampling_ratio/max": 1.9989465475082397, "entropy": 0.14800493232905865, "clip_ratio/low_mean": 0.013456470100209117, "clip_ratio/low_min": 0.013456470100209117, "clip_ratio/high_mean": 0.0018382353009656072, "clip_ratio/high_max": 0.0018382353009656072, "clip_ratio/region_mean": 0.015294705401174724, "reward_total_mean": 0.9620155096054077, "reward_meter_mean": 0.9620155096054077, "reward_meter_std": 0.010956778191030025, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9620155096054077, "reward_total_composite_std": 0.010956778191030025} {"timestamp_utc": "2026-04-12T01:42:17Z", "mode": "train", "global_step": 2245, "epoch": 0.09017150660722176, "loss": 0.0006, "grad_norm": 1.3484723567962646, "learning_rate": 3.2000000000000003e-06, "num_tokens": 5076689.0, "completions/mean_length": 167.0, "completions/min_length": 167.0, "completions/max_length": 167.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 167.0, "completions/min_terminated_length": 167.0, "completions/max_terminated_length": 167.0, "rewards/meter/mean": 0.9959465265274048, "rewards/meter/std": 9.931746171787381e-05, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8409091234207153, "rewards/repeat_penalty/std": 0.08058230578899384, "rewards/total_composite/mean": 0.717853307723999, "rewards/total_composite/std": 0.06874335557222366, "reward": 0.717853307723999, "reward_std": 0.06874334812164307, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010420297272503376, "sampling/sampling_logp_difference/max": 1.2074825763702393, "sampling/importance_sampling_ratio/min": 0.29894891381263733, "sampling/importance_sampling_ratio/mean": 0.9994251132011414, "sampling/importance_sampling_ratio/max": 1.3406785726547241, "entropy": 0.05933140218257904, "clip_ratio/low_mean": 0.0022455090656876564, "clip_ratio/low_min": 0.0022455090656876564, "clip_ratio/high_mean": 0.005239521153271198, "clip_ratio/high_max": 0.005239521153271198, "clip_ratio/region_mean": 0.007485030218958855, "reward_total_mean": 0.717853307723999, "reward_meter_mean": 0.9959465265274048, "reward_meter_std": 9.931746171787381e-05, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8409091234207153, "reward_repeat_penalty_std": 0.08058230578899384, "reward_total_composite_mean": 0.717853307723999, "reward_total_composite_std": 0.06874335557222366} {"timestamp_utc": "2026-04-12T01:42:23Z", "mode": "train", "global_step": 2246, "epoch": 0.09021167208900671, "loss": 0.0044, "grad_norm": 2.09184193611145, "learning_rate": 3.196969696969697e-06, "num_tokens": 5079477.0, "completions/mean_length": 143.5, "completions/min_length": 136.0, "completions/max_length": 164.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 143.5, "completions/min_terminated_length": 136.0, "completions/max_terminated_length": 164.0, "rewards/meter/mean": 0.9908416867256165, "rewards/meter/std": 0.005638771690428257, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8392857313156128, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.6652302742004395, "rewards/total_composite/std": 0.03939948230981827, "reward": 0.6652302742004395, "reward_std": 0.03939947113394737, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02601451985538006, "sampling/sampling_logp_difference/max": 1.5025901794433594, "sampling/importance_sampling_ratio/min": 0.22255297005176544, "sampling/importance_sampling_ratio/mean": 1.004660725593567, "sampling/importance_sampling_ratio/max": 1.8299243450164795, "entropy": 0.19038555212318897, "clip_ratio/low_mean": 0.0008802816737443209, "clip_ratio/low_min": 0.0008802816737443209, "clip_ratio/high_mean": 0.018073388375341892, "clip_ratio/high_max": 0.018073388375341892, "clip_ratio/region_mean": 0.018953670049086213, "reward_total_mean": 0.6652302742004395, "reward_meter_mean": 0.9908416867256165, "reward_meter_std": 0.005638771690428257, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8392857313156128, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.6652302742004395, "reward_total_composite_std": 0.03939948230981827} {"timestamp_utc": "2026-04-12T01:42:29Z", "mode": "train", "global_step": 2247, "epoch": 0.09025183757079167, "loss": 0.0006, "grad_norm": 2.32881760597229, "learning_rate": 3.1939393939393944e-06, "num_tokens": 5082066.0, "completions/mean_length": 142.625, "completions/min_length": 141.0, "completions/max_length": 144.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 142.625, "completions/min_terminated_length": 141.0, "completions/max_terminated_length": 144.0, "rewards/meter/mean": 0.9939021468162537, "rewards/meter/std": 0.0005415793275460601, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8519160747528076, "rewards/total_composite/std": 0.0004642106650862843, "reward": 0.8519160747528076, "reward_std": 0.0004642065614461899, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014642242342233658, "sampling/sampling_logp_difference/max": 1.4676084518432617, "sampling/importance_sampling_ratio/min": 0.2304760217666626, "sampling/importance_sampling_ratio/mean": 1.001724362373352, "sampling/importance_sampling_ratio/max": 1.7263494729995728, "entropy": 0.10466479323804379, "clip_ratio/low_mean": 0.001760650658980012, "clip_ratio/low_min": 0.001760650658980012, "clip_ratio/high_mean": 0.0008741258643567562, "clip_ratio/high_max": 0.0008741258643567562, "clip_ratio/region_mean": 0.002634776523336768, "reward_total_mean": 0.8519160747528076, "reward_meter_mean": 0.9939021468162537, "reward_meter_std": 0.0005415793275460601, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8519160747528076, "reward_total_composite_std": 0.0004642106650862843} {"timestamp_utc": "2026-04-12T01:42:34Z", "mode": "train", "global_step": 2248, "epoch": 0.09029200305257662, "loss": 0.0011, "grad_norm": 2.4443159103393555, "learning_rate": 3.190909090909091e-06, "num_tokens": 5084426.0, "completions/mean_length": 111.0, "completions/min_length": 111.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 111.0, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.9956398010253906, "rewards/meter/std": 9.625943494029343e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9422965049743652, "rewards/total_composite/std": 0.07353661954402924, "reward": 0.9422965049743652, "reward_std": 0.07353662699460983, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008148727007210255, "sampling/sampling_logp_difference/max": 0.5144362449645996, "sampling/importance_sampling_ratio/min": 0.5978375673294067, "sampling/importance_sampling_ratio/mean": 1.0003148317337036, "sampling/importance_sampling_ratio/max": 1.3495516777038574, "entropy": 0.06026614410802722, "clip_ratio/low_mean": 0.0022522523067891598, "clip_ratio/low_min": 0.0022522523067891598, "clip_ratio/high_mean": 0.0045045046135783195, "clip_ratio/high_max": 0.0045045046135783195, "clip_ratio/region_mean": 0.006756756920367479, "reward_total_mean": 0.9422965049743652, "reward_meter_mean": 0.9956398010253906, "reward_meter_std": 9.625943494029343e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.9422965049743652, "reward_total_composite_std": 0.07353661954402924} {"timestamp_utc": "2026-04-12T01:42:38Z", "mode": "train", "global_step": 2249, "epoch": 0.09033216853436157, "loss": 0.0001, "grad_norm": 0.3145919442176819, "learning_rate": 3.187878787878788e-06, "num_tokens": 5086546.0, "completions/mean_length": 69.0, "completions/min_length": 69.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.0, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9975405931472778, "rewards/meter/std": 1.4270765859691892e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9975405931472778, "rewards/total_composite/std": 1.4270765859691892e-05, "reward": 0.9975405931472778, "reward_std": 1.4251797438191716e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006101167760789394, "sampling/sampling_logp_difference/max": 0.688692569732666, "sampling/importance_sampling_ratio/min": 0.5022322535514832, "sampling/importance_sampling_ratio/mean": 1.0008234977722168, "sampling/importance_sampling_ratio/max": 1.2142425775527954, "entropy": 0.05507086403667927, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9975405931472778, "reward_meter_mean": 0.9975405931472778, "reward_meter_std": 1.4270765859691892e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9975405931472778, "reward_total_composite_std": 1.4270765859691892e-05} {"timestamp_utc": "2026-04-12T01:42:43Z", "mode": "train", "global_step": 2250, "epoch": 0.09037233401614653, "loss": -0.0011, "grad_norm": 0.5331827402114868, "learning_rate": 3.1848484848484853e-06, "num_tokens": 5088505.0, "completions/mean_length": 68.875, "completions/min_length": 68.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.875, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9975307583808899, "rewards/meter/std": 2.9643033485626802e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9975307583808899, "rewards/total_composite/std": 2.9643033485626802e-05, "reward": 0.9975307583808899, "reward_std": 2.9639279091497883e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005882658995687962, "sampling/sampling_logp_difference/max": 0.6369953155517578, "sampling/importance_sampling_ratio/min": 0.5288791656494141, "sampling/importance_sampling_ratio/mean": 1.0018889904022217, "sampling/importance_sampling_ratio/max": 1.2806227207183838, "entropy": 0.0506328740157187, "clip_ratio/low_mean": 0.0036498295376077294, "clip_ratio/low_min": 0.0036498295376077294, "clip_ratio/high_mean": 0.0036231884732842445, "clip_ratio/high_max": 0.0036231884732842445, "clip_ratio/region_mean": 0.007273018010891974, "reward_total_mean": 0.9975307583808899, "reward_meter_mean": 0.9975307583808899, "reward_meter_std": 2.9643033485626802e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9975307583808899, "reward_total_composite_std": 2.9643033485626802e-05} {"timestamp_utc": "2026-04-12T01:43:47Z", "mode": "eval", "global_step": 2250, "epoch": 0.09037233401614653, "eval_loss": NaN, "eval_runtime": 64.292, "eval_samples_per_second": 1.618, "eval_steps_per_second": 0.202, "eval_num_tokens": 5088505.0, "eval_completions/mean_length": 182.43269230769232, "eval_completions/min_length": 59.53846153846154, "eval_completions/max_length": 330.53846153846155, "eval_completions/clipped_ratio": 0.009615384615384616, "eval_completions/mean_terminated_length": 180.01648418719952, "eval_completions/min_terminated_length": 59.53846153846154, "eval_completions/max_terminated_length": 319.53846153846155, "eval_rewards/meter/mean": 0.732613939505357, "eval_rewards/meter/std": 0.3826414684836681, "eval_rewards/count_adherence/mean": 0.8588019150954026, "eval_rewards/count_adherence/std": 0.1499052156622593, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/repeat_penalty/mean": 0.875946778517503, "eval_rewards/repeat_penalty/std": 0.12293276706567177, "eval_rewards/total_composite/mean": 0.5561339992743272, "eval_rewards/total_composite/std": 0.3474527712051685, "eval_reward": 0.5561339992743272, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.0185515835451392, "eval_sampling/sampling_logp_difference/max": 0.9445312756758469, "eval_sampling/importance_sampling_ratio/min": 0.40484776290563435, "eval_sampling/importance_sampling_ratio/mean": 1.0042996865052443, "eval_sampling/importance_sampling_ratio/max": 1.410868342106159, "eval_entropy": 0.19319924941429725, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5561339992743272, "eval_reward_meter_mean": 0.732613939505357, "eval_reward_meter_std": 0.3826414684836681, "eval_reward_count_adherence_mean": 0.8588019150954026, "eval_reward_count_adherence_std": 0.1499052156622593, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_repeat_penalty_mean": 0.875946778517503, "eval_reward_repeat_penalty_std": 0.12293276706567177, "eval_reward_total_composite_mean": 0.5561339992743272, "eval_reward_total_composite_std": 0.3474527712051685} {"timestamp_utc": "2026-04-12T01:43:55Z", "mode": "train", "global_step": 2251, "epoch": 0.09041249949793148, "loss": 0.0069, "grad_norm": 2.5881564617156982, "learning_rate": 3.181818181818182e-06, "num_tokens": 5090737.0, "completions/mean_length": 102.0, "completions/min_length": 101.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.0, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.9657948613166809, "rewards/meter/std": 0.005292544141411781, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9657948613166809, "rewards/total_composite/std": 0.005292544141411781, "reward": 0.9657948613166809, "reward_std": 0.005292540416121483, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022372471168637276, "sampling/sampling_logp_difference/max": 1.110783576965332, "sampling/importance_sampling_ratio/min": 0.32930082082748413, "sampling/importance_sampling_ratio/mean": 1.007267713546753, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15675346460193396, "clip_ratio/low_mean": 0.009804157423786819, "clip_ratio/low_min": 0.009804157423786819, "clip_ratio/high_mean": 0.003688839729875326, "clip_ratio/high_max": 0.003688839729875326, "clip_ratio/region_mean": 0.013492997153662145, "reward_total_mean": 0.9657948613166809, "reward_meter_mean": 0.9657948613166809, "reward_meter_std": 0.005292544141411781, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9657948613166809, "reward_total_composite_std": 0.005292544141411781} {"timestamp_utc": "2026-04-12T01:44:00Z", "mode": "train", "global_step": 2252, "epoch": 0.09045266497971644, "loss": 0.0092, "grad_norm": 8.668261528015137, "learning_rate": 3.178787878787879e-06, "num_tokens": 5092817.0, "completions/mean_length": 71.0, "completions/min_length": 69.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.0, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9977865219116211, "rewards/meter/std": 0.0025578902568668127, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977865219116211, "rewards/total_composite/std": 0.0025578902568668127, "reward": 0.9977865219116211, "reward_std": 0.0025578862987458706, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06398905813694, "sampling/sampling_logp_difference/max": 1.5597095489501953, "sampling/importance_sampling_ratio/min": 0.2101971060037613, "sampling/importance_sampling_ratio/mean": 1.0122182369232178, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43995901197195053, "clip_ratio/low_mean": 0.012399396393448114, "clip_ratio/low_min": 0.012399396393448114, "clip_ratio/high_mean": 0.031607592245563865, "clip_ratio/high_max": 0.031607592245563865, "clip_ratio/region_mean": 0.04400698863901198, "reward_total_mean": 0.9977865219116211, "reward_meter_mean": 0.9977865219116211, "reward_meter_std": 0.0025578902568668127, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9977865219116211, "reward_total_composite_std": 0.0025578902568668127} {"timestamp_utc": "2026-04-12T01:44:04Z", "mode": "train", "global_step": 2253, "epoch": 0.09049283046150139, "loss": -0.0002, "grad_norm": 0.0889202132821083, "learning_rate": 3.1757575757575758e-06, "num_tokens": 5094681.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.7876288890838623, "rewards/meter/std": 2.8027659936924465e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7876288890838623, "rewards/total_composite/std": 2.8027659936924465e-05, "reward": 0.7876288890838623, "reward_std": 2.8021637263009325e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0013179951347410679, "sampling/sampling_logp_difference/max": 0.07088017463684082, "sampling/importance_sampling_ratio/min": 0.9315734505653381, "sampling/importance_sampling_ratio/mean": 1.0006846189498901, "sampling/importance_sampling_ratio/max": 1.052701711654663, "entropy": 0.01149469253141433, "clip_ratio/low_mean": 0.002314814832061529, "clip_ratio/low_min": 0.002314814832061529, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.002314814832061529, "reward_total_mean": 0.7876288890838623, "reward_meter_mean": 0.7876288890838623, "reward_meter_std": 2.8027659936924465e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7876288890838623, "reward_total_composite_std": 2.8027659936924465e-05} {"timestamp_utc": "2026-04-12T01:44:11Z", "mode": "train", "global_step": 2254, "epoch": 0.09053299594328634, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.172727272727273e-06, "num_tokens": 5096369.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0015978036681190133, "sampling/sampling_logp_difference/max": 0.11908148974180222, "sampling/importance_sampling_ratio/min": 0.8877354264259338, "sampling/importance_sampling_ratio/mean": 1.0007966756820679, "sampling/importance_sampling_ratio/max": 1.057936429977417, "entropy": 0.010870376718230546, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:44:15Z", "mode": "train", "global_step": 2255, "epoch": 0.0905731614250713, "loss": -0.0042, "grad_norm": 11.48292350769043, "learning_rate": 3.16969696969697e-06, "num_tokens": 5097900.0, "completions/mean_length": 33.375, "completions/min_length": 33.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.375, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9692210555076599, "rewards/meter/std": 0.003493846394121647, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9692210555076599, "rewards/total_composite/std": 0.003493846394121647, "reward": 0.9692210555076599, "reward_std": 0.0034938445314764977, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03685401380062103, "sampling/sampling_logp_difference/max": 1.65773606300354, "sampling/importance_sampling_ratio/min": 0.1905699372291565, "sampling/importance_sampling_ratio/mean": 1.0003447532653809, "sampling/importance_sampling_ratio/max": 1.707613468170166, "entropy": 0.1904709991067648, "clip_ratio/low_mean": 0.007464349502697587, "clip_ratio/low_min": 0.007464349502697587, "clip_ratio/high_mean": 0.018939394503831863, "clip_ratio/high_max": 0.018939394503831863, "clip_ratio/region_mean": 0.02640374400652945, "reward_total_mean": 0.9692210555076599, "reward_meter_mean": 0.9692210555076599, "reward_meter_std": 0.003493846394121647, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9692210555076599, "reward_total_composite_std": 0.003493846394121647} {"timestamp_utc": "2026-04-12T01:44:21Z", "mode": "train", "global_step": 2256, "epoch": 0.09061332690685625, "loss": 0.0115, "grad_norm": 2.849970817565918, "learning_rate": 3.1666666666666667e-06, "num_tokens": 5100800.0, "completions/mean_length": 176.5, "completions/min_length": 169.0, "completions/max_length": 188.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 176.5, "completions/min_terminated_length": 169.0, "completions/max_terminated_length": 188.0, "rewards/meter/mean": 0.998549222946167, "rewards/meter/std": 0.00034224658156745136, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9027777910232544, "rewards/repeat_penalty/std": 0.07120776921510696, "rewards/total_composite/mean": 0.7512168884277344, "rewards/total_composite/std": 0.05916035175323486, "reward": 0.7512168884277344, "reward_std": 0.05916035920381546, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04010716825723648, "sampling/sampling_logp_difference/max": 2.1722252368927, "sampling/importance_sampling_ratio/min": 0.1139238253235817, "sampling/importance_sampling_ratio/mean": 1.005678415298462, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.27090173214673996, "clip_ratio/low_mean": 0.02541261224541813, "clip_ratio/low_min": 0.02541261224541813, "clip_ratio/high_mean": 0.011609932873398066, "clip_ratio/high_max": 0.011609932873398066, "clip_ratio/region_mean": 0.0370225451188162, "reward_total_mean": 0.7512168884277344, "reward_meter_mean": 0.998549222946167, "reward_meter_std": 0.00034224658156745136, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9027777910232544, "reward_repeat_penalty_std": 0.07120776921510696, "reward_total_composite_mean": 0.7512168884277344, "reward_total_composite_std": 0.05916035175323486} {"timestamp_utc": "2026-04-12T01:44:28Z", "mode": "train", "global_step": 2257, "epoch": 0.0906534923886412, "loss": -0.0557, "grad_norm": 2.3163866996765137, "learning_rate": 3.1636363636363635e-06, "num_tokens": 5104173.0, "completions/mean_length": 222.625, "completions/min_length": 192.0, "completions/max_length": 244.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 222.625, "completions/min_terminated_length": 192.0, "completions/max_terminated_length": 244.0, "rewards/meter/mean": 0.9975998401641846, "rewards/meter/std": 0.0006318566738627851, "rewards/count_adherence/mean": 0.8035714030265808, "rewards/count_adherence/std": 0.07393559068441391, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8901515007019043, "rewards/repeat_penalty/std": 0.057505469769239426, "rewards/total_composite/mean": 0.7136683464050293, "rewards/total_composite/std": 0.08285731077194214, "reward": 0.7136683464050293, "reward_std": 0.08285731077194214, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03854899853467941, "sampling/sampling_logp_difference/max": 1.9127869606018066, "sampling/importance_sampling_ratio/min": 0.14766825735569, "sampling/importance_sampling_ratio/mean": 1.0057517290115356, "sampling/importance_sampling_ratio/max": 1.9951218366622925, "entropy": 0.273501418530941, "clip_ratio/low_mean": 0.01093733258312568, "clip_ratio/low_min": 0.01093733258312568, "clip_ratio/high_mean": 0.01619163854047656, "clip_ratio/high_max": 0.01619163854047656, "clip_ratio/region_mean": 0.02712897112360224, "reward_total_mean": 0.7136683464050293, "reward_meter_mean": 0.9975998401641846, "reward_meter_std": 0.0006318566738627851, "reward_count_adherence_mean": 0.8035714030265808, "reward_count_adherence_std": 0.07393559068441391, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8901515007019043, "reward_repeat_penalty_std": 0.057505469769239426, "reward_total_composite_mean": 0.7136683464050293, "reward_total_composite_std": 0.08285731077194214} {"timestamp_utc": "2026-04-12T01:44:34Z", "mode": "train", "global_step": 2258, "epoch": 0.09069365787042616, "loss": 0.014, "grad_norm": 2.4006783962249756, "learning_rate": 3.1606060606060608e-06, "num_tokens": 5106939.0, "completions/mean_length": 173.75, "completions/min_length": 170.0, "completions/max_length": 177.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 173.75, "completions/min_terminated_length": 170.0, "completions/max_terminated_length": 177.0, "rewards/meter/mean": 0.9978283047676086, "rewards/meter/std": 0.0015693600289523602, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333134651184, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.7968448400497437, "rewards/total_composite/std": 0.04724044352769852, "reward": 0.7968448400497437, "reward_std": 0.04724043607711792, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03900068625807762, "sampling/sampling_logp_difference/max": 1.3687856197357178, "sampling/importance_sampling_ratio/min": 0.25441572070121765, "sampling/importance_sampling_ratio/mean": 1.0058785676956177, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.29748900793492794, "clip_ratio/low_mean": 0.00781669607385993, "clip_ratio/low_min": 0.00781669607385993, "clip_ratio/high_mean": 0.029055888997390866, "clip_ratio/high_max": 0.029055888997390866, "clip_ratio/region_mean": 0.036872585071250796, "reward_total_mean": 0.7968448400497437, "reward_meter_mean": 0.9978283047676086, "reward_meter_std": 0.0015693600289523602, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333134651184, "reward_repeat_penalty_std": 0.05750546231865883, "reward_total_composite_mean": 0.7968448400497437, "reward_total_composite_std": 0.04724044352769852} {"timestamp_utc": "2026-04-12T01:44:39Z", "mode": "train", "global_step": 2259, "epoch": 0.09073382335221111, "loss": -0.0036, "grad_norm": 5.68310022354126, "learning_rate": 3.1575757575757576e-06, "num_tokens": 5108781.0, "completions/mean_length": 70.25, "completions/min_length": 67.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.25, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9816989898681641, "rewards/meter/std": 0.026165146380662918, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9816989898681641, "rewards/total_composite/std": 0.026165146380662918, "reward": 0.9816989898681641, "reward_std": 0.02616514265537262, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031211039051413536, "sampling/sampling_logp_difference/max": 1.0319323539733887, "sampling/importance_sampling_ratio/min": 0.35631778836250305, "sampling/importance_sampling_ratio/mean": 1.0080952644348145, "sampling/importance_sampling_ratio/max": 1.8370952606201172, "entropy": 0.27262550778687, "clip_ratio/low_mean": 0.0035714285913854837, "clip_ratio/low_min": 0.0035714285913854837, "clip_ratio/high_mean": 0.024707219563424587, "clip_ratio/high_max": 0.024707219563424587, "clip_ratio/region_mean": 0.02827864815481007, "reward_total_mean": 0.9816989898681641, "reward_meter_mean": 0.9816989898681641, "reward_meter_std": 0.026165146380662918, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9816989898681641, "reward_total_composite_std": 0.026165146380662918} {"timestamp_utc": "2026-04-12T01:44:47Z", "mode": "train", "global_step": 2260, "epoch": 0.09077398883399607, "loss": -0.0005, "grad_norm": 1.6112135648727417, "learning_rate": 3.1545454545454545e-06, "num_tokens": 5113402.0, "completions/mean_length": 355.625, "completions/min_length": 346.0, "completions/max_length": 369.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 355.625, "completions/min_terminated_length": 346.0, "completions/max_terminated_length": 369.0, "rewards/meter/mean": 0.9965711236000061, "rewards/meter/std": 0.0014478074153885245, "rewards/count_adherence/mean": 0.6428571343421936, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8088235259056091, "rewards/repeat_penalty/std": 0.07539646327495575, "rewards/total_composite/mean": 0.5181742906570435, "rewards/total_composite/std": 0.04829566553235054, "reward": 0.5181742906570435, "reward_std": 0.048295676708221436, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.034783411771059036, "sampling/sampling_logp_difference/max": 2.516125440597534, "sampling/importance_sampling_ratio/min": 0.08077196031808853, "sampling/importance_sampling_ratio/mean": 1.0037370920181274, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2432896252721548, "clip_ratio/low_mean": 0.005656401859596372, "clip_ratio/low_min": 0.005656401859596372, "clip_ratio/high_mean": 0.0209878021851182, "clip_ratio/high_max": 0.0209878021851182, "clip_ratio/region_mean": 0.02664420404471457, "reward_total_mean": 0.5181742906570435, "reward_meter_mean": 0.9965711236000061, "reward_meter_std": 0.0014478074153885245, "reward_count_adherence_mean": 0.6428571343421936, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8088235259056091, "reward_repeat_penalty_std": 0.07539646327495575, "reward_total_composite_mean": 0.5181742906570435, "reward_total_composite_std": 0.04829566553235054} {"timestamp_utc": "2026-04-12T01:44:53Z", "mode": "train", "global_step": 2261, "epoch": 0.09081415431578102, "loss": 0.0152, "grad_norm": 3.7612602710723877, "learning_rate": 3.1515151515151517e-06, "num_tokens": 5115515.0, "completions/mean_length": 104.125, "completions/min_length": 100.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 104.125, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.9984878301620483, "rewards/meter/std": 0.000643216073513031, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9735202789306641, "rewards/total_composite/std": 0.07053547352552414, "reward": 0.9735202789306641, "reward_std": 0.07053548842668533, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04828030988574028, "sampling/sampling_logp_difference/max": 2.574674606323242, "sampling/importance_sampling_ratio/min": 0.07617860287427902, "sampling/importance_sampling_ratio/mean": 1.0051672458648682, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.34618273936212063, "clip_ratio/low_mean": 0.00589622650295496, "clip_ratio/low_min": 0.00589622650295496, "clip_ratio/high_mean": 0.04133175266906619, "clip_ratio/high_max": 0.04133175266906619, "clip_ratio/region_mean": 0.04722797917202115, "reward_total_mean": 0.9735202789306641, "reward_meter_mean": 0.9984878301620483, "reward_meter_std": 0.000643216073513031, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9735202789306641, "reward_total_composite_std": 0.07053547352552414} {"timestamp_utc": "2026-04-12T01:44:57Z", "mode": "train", "global_step": 2262, "epoch": 0.09085431979756597, "loss": 0.0082, "grad_norm": 3.4450576305389404, "learning_rate": 3.1484848484848485e-06, "num_tokens": 5117374.0, "completions/mean_length": 68.375, "completions/min_length": 67.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.375, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9669647812843323, "rewards/meter/std": 0.014694334007799625, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9669647812843323, "rewards/total_composite/std": 0.014694334007799625, "reward": 0.9669647812843323, "reward_std": 0.014694339595735073, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028783582150936127, "sampling/sampling_logp_difference/max": 1.1841890811920166, "sampling/importance_sampling_ratio/min": 0.3059942126274109, "sampling/importance_sampling_ratio/mean": 1.0112721920013428, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.19510209374129772, "clip_ratio/low_mean": 0.0018115942366421223, "clip_ratio/low_min": 0.0018115942366421223, "clip_ratio/high_mean": 0.0202221788931638, "clip_ratio/high_max": 0.0202221788931638, "clip_ratio/region_mean": 0.022033773129805923, "reward_total_mean": 0.9669647812843323, "reward_meter_mean": 0.9669647812843323, "reward_meter_std": 0.014694334007799625, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9669647812843323, "reward_total_composite_std": 0.014694334007799625} {"timestamp_utc": "2026-04-12T01:45:01Z", "mode": "train", "global_step": 2263, "epoch": 0.09089448527935093, "loss": 0.0034, "grad_norm": 1.744556188583374, "learning_rate": 3.145454545454546e-06, "num_tokens": 5118823.0, "completions/mean_length": 32.125, "completions/min_length": 32.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.125, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9992284774780273, "rewards/meter/std": 0.00014189988723956048, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992284774780273, "rewards/total_composite/std": 0.00014189988723956048, "reward": 0.9992284774780273, "reward_std": 0.00014191023365128785, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005864746868610382, "sampling/sampling_logp_difference/max": 0.17466145753860474, "sampling/importance_sampling_ratio/min": 0.8397412300109863, "sampling/importance_sampling_ratio/mean": 0.9978591203689575, "sampling/importance_sampling_ratio/max": 1.1715004444122314, "entropy": 0.05214253021404147, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/high_mean": 0.00390625, "clip_ratio/high_max": 0.00390625, "clip_ratio/region_mean": 0.007694128900766373, "reward_total_mean": 0.9992284774780273, "reward_meter_mean": 0.9992284774780273, "reward_meter_std": 0.00014189988723956048, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992284774780273, "reward_total_composite_std": 0.00014189988723956048} {"timestamp_utc": "2026-04-12T01:45:06Z", "mode": "train", "global_step": 2264, "epoch": 0.09093465076113588, "loss": -0.0007, "grad_norm": 4.9907097816467285, "learning_rate": 3.142424242424243e-06, "num_tokens": 5120274.0, "completions/mean_length": 33.375, "completions/min_length": 33.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.375, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9775052070617676, "rewards/meter/std": 0.010550287552177906, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9775052070617676, "rewards/total_composite/std": 0.010550287552177906, "reward": 0.9775052070617676, "reward_std": 0.010550293140113354, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02235337160527706, "sampling/sampling_logp_difference/max": 0.6383800506591797, "sampling/importance_sampling_ratio/min": 0.528147280216217, "sampling/importance_sampling_ratio/mean": 1.0072755813598633, "sampling/importance_sampling_ratio/max": 1.8347361087799072, "entropy": 0.16098480112850666, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/high_mean": 0.014823656994849443, "clip_ratio/high_max": 0.014823656994849443, "clip_ratio/region_mean": 0.018611535895615816, "reward_total_mean": 0.9775052070617676, "reward_meter_mean": 0.9775052070617676, "reward_meter_std": 0.010550287552177906, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9775052070617676, "reward_total_composite_std": 0.010550287552177906} {"timestamp_utc": "2026-04-12T01:45:11Z", "mode": "train", "global_step": 2265, "epoch": 0.09097481624292084, "loss": 0.0095, "grad_norm": 2.8727030754089355, "learning_rate": 3.13939393939394e-06, "num_tokens": 5122367.0, "completions/mean_length": 83.625, "completions/min_length": 83.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.625, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.9946054816246033, "rewards/meter/std": 0.0010328061180189252, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9946054816246033, "rewards/total_composite/std": 0.0010328061180189252, "reward": 0.9946054816246033, "reward_std": 0.0010328061180189252, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018597107380628586, "sampling/sampling_logp_difference/max": 1.4482271671295166, "sampling/importance_sampling_ratio/min": 0.23498652875423431, "sampling/importance_sampling_ratio/mean": 1.0014218091964722, "sampling/importance_sampling_ratio/max": 1.7508782148361206, "entropy": 0.08839869406074286, "clip_ratio/low_mean": 0.01339285762514919, "clip_ratio/low_min": 0.01339285762514919, "clip_ratio/high_mean": 0.00745840510353446, "clip_ratio/high_max": 0.00745840510353446, "clip_ratio/region_mean": 0.02085126272868365, "reward_total_mean": 0.9946054816246033, "reward_meter_mean": 0.9946054816246033, "reward_meter_std": 0.0010328061180189252, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9946054816246033, "reward_total_composite_std": 0.0010328061180189252} {"timestamp_utc": "2026-04-12T01:45:16Z", "mode": "train", "global_step": 2266, "epoch": 0.09101498172470579, "loss": 0.0001, "grad_norm": 3.1527485847473145, "learning_rate": 3.1363636363636367e-06, "num_tokens": 5125064.0, "completions/mean_length": 140.125, "completions/min_length": 139.0, "completions/max_length": 143.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 140.125, "completions/min_terminated_length": 139.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.9949551820755005, "rewards/meter/std": 0.0007479682681150734, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.819444477558136, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.81532883644104, "rewards/total_composite/std": 0.05752642825245857, "reward": 0.81532883644104, "reward_std": 0.057526424527168274, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01804959587752819, "sampling/sampling_logp_difference/max": 1.1149301528930664, "sampling/importance_sampling_ratio/min": 0.32793816924095154, "sampling/importance_sampling_ratio/mean": 1.0007151365280151, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09540552459657192, "clip_ratio/low_mean": 0.006256514578126371, "clip_ratio/low_min": 0.006256514578126371, "clip_ratio/high_mean": 0.007080778945237398, "clip_ratio/high_max": 0.007080778945237398, "clip_ratio/region_mean": 0.013337293523363769, "reward_total_mean": 0.81532883644104, "reward_meter_mean": 0.9949551820755005, "reward_meter_std": 0.0007479682681150734, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.819444477558136, "reward_repeat_penalty_std": 0.05750546231865883, "reward_total_composite_mean": 0.81532883644104, "reward_total_composite_std": 0.05752642825245857} {"timestamp_utc": "2026-04-12T01:45:22Z", "mode": "train", "global_step": 2267, "epoch": 0.09105514720649074, "loss": 0.0082, "grad_norm": 2.1978232860565186, "learning_rate": 3.133333333333334e-06, "num_tokens": 5127489.0, "completions/mean_length": 135.125, "completions/min_length": 134.0, "completions/max_length": 137.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 135.125, "completions/min_terminated_length": 134.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.9982975721359253, "rewards/meter/std": 0.000450806604931131, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9982975721359253, "rewards/total_composite/std": 0.000450806604931131, "reward": 0.9982975721359253, "reward_std": 0.00045079842675477266, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022603249177336693, "sampling/sampling_logp_difference/max": 1.3583526611328125, "sampling/importance_sampling_ratio/min": 0.2570839524269104, "sampling/importance_sampling_ratio/mean": 0.9998656511306763, "sampling/importance_sampling_ratio/max": 1.8831219673156738, "entropy": 0.09932332392781973, "clip_ratio/low_mean": 0.006448142405133694, "clip_ratio/low_min": 0.006448142405133694, "clip_ratio/high_mean": 0.012057766725774854, "clip_ratio/high_max": 0.012057766725774854, "clip_ratio/region_mean": 0.01850590913090855, "reward_total_mean": 0.9982975721359253, "reward_meter_mean": 0.9982975721359253, "reward_meter_std": 0.000450806604931131, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9982975721359253, "reward_total_composite_std": 0.000450806604931131} {"timestamp_utc": "2026-04-12T01:45:27Z", "mode": "train", "global_step": 2268, "epoch": 0.0910953126882757, "loss": 0.0066, "grad_norm": 3.012678384780884, "learning_rate": 3.130303030303031e-06, "num_tokens": 5130007.0, "completions/mean_length": 135.75, "completions/min_length": 132.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 135.75, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.9679410457611084, "rewards/meter/std": 0.006429874338209629, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.8470007181167603, "rewards/total_composite/std": 0.05034321919083595, "reward": 0.8470007181167603, "reward_std": 0.05034322291612625, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022967170923948288, "sampling/sampling_logp_difference/max": 1.6587629318237305, "sampling/importance_sampling_ratio/min": 0.3170107305049896, "sampling/importance_sampling_ratio/mean": 1.0043870210647583, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1438667206093669, "clip_ratio/low_mean": 0.013712125248275697, "clip_ratio/low_min": 0.013712125248275697, "clip_ratio/high_mean": 0.005474452394992113, "clip_ratio/high_max": 0.005474452394992113, "clip_ratio/region_mean": 0.01918657764326781, "reward_total_mean": 0.8470007181167603, "reward_meter_mean": 0.9679410457611084, "reward_meter_std": 0.006429874338209629, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.8470007181167603, "reward_total_composite_std": 0.05034321919083595} {"timestamp_utc": "2026-04-12T01:45:33Z", "mode": "train", "global_step": 2269, "epoch": 0.09113547817006065, "loss": 0.0004, "grad_norm": 2.066251754760742, "learning_rate": 3.1272727272727276e-06, "num_tokens": 5133013.0, "completions/mean_length": 178.75, "completions/min_length": 176.0, "completions/max_length": 181.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 178.75, "completions/min_terminated_length": 176.0, "completions/max_terminated_length": 181.0, "rewards/meter/mean": 0.9990208148956299, "rewards/meter/std": 0.00016956882609520108, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8888888955116272, "rewards/repeat_penalty/std": 0.10286889225244522, "rewards/total_composite/mean": 0.7400156855583191, "rewards/total_composite/std": 0.08564041554927826, "reward": 0.7400156855583191, "reward_std": 0.08564040064811707, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023759307339787483, "sampling/sampling_logp_difference/max": 1.466665267944336, "sampling/importance_sampling_ratio/min": 0.2306935042142868, "sampling/importance_sampling_ratio/mean": 1.0039939880371094, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.19854111224412918, "clip_ratio/low_mean": 0.0028210257878527045, "clip_ratio/low_min": 0.0028210257878527045, "clip_ratio/high_mean": 0.012555458233691752, "clip_ratio/high_max": 0.012555458233691752, "clip_ratio/region_mean": 0.015376484021544456, "reward_total_mean": 0.7400156855583191, "reward_meter_mean": 0.9990208148956299, "reward_meter_std": 0.00016956882609520108, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8888888955116272, "reward_repeat_penalty_std": 0.10286889225244522, "reward_total_composite_mean": 0.7400156855583191, "reward_total_composite_std": 0.08564041554927826} {"timestamp_utc": "2026-04-12T01:45:39Z", "mode": "train", "global_step": 2270, "epoch": 0.0911756436518456, "loss": 0.0089, "grad_norm": 3.698033094406128, "learning_rate": 3.1242424242424245e-06, "num_tokens": 5136127.0, "completions/mean_length": 170.25, "completions/min_length": 162.0, "completions/max_length": 177.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 170.25, "completions/min_terminated_length": 162.0, "completions/max_terminated_length": 177.0, "rewards/meter/mean": 0.9971339702606201, "rewards/meter/std": 0.0020925395656377077, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333134651184, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.7962738275527954, "rewards/total_composite/std": 0.04691069945693016, "reward": 0.7962738275527954, "reward_std": 0.04691069573163986, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0504683218896389, "sampling/sampling_logp_difference/max": 2.827153444290161, "sampling/importance_sampling_ratio/min": 0.05918107181787491, "sampling/importance_sampling_ratio/mean": 1.011202096939087, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.38951336964964867, "clip_ratio/low_mean": 0.01330830343067646, "clip_ratio/low_min": 0.01330830343067646, "clip_ratio/high_mean": 0.02643707417882979, "clip_ratio/high_max": 0.02643707417882979, "clip_ratio/region_mean": 0.03974537760950625, "reward_total_mean": 0.7962738275527954, "reward_meter_mean": 0.9971339702606201, "reward_meter_std": 0.0020925395656377077, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333134651184, "reward_repeat_penalty_std": 0.05750546231865883, "reward_total_composite_mean": 0.7962738275527954, "reward_total_composite_std": 0.04691069945693016} {"timestamp_utc": "2026-04-12T01:45:44Z", "mode": "train", "global_step": 2271, "epoch": 0.09121580913363056, "loss": 0.0327, "grad_norm": 6.407037258148193, "learning_rate": 3.1212121212121217e-06, "num_tokens": 5137976.0, "completions/mean_length": 62.125, "completions/min_length": 57.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.125, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.09545363485813141, "rewards/meter/std": 0.17621159553527832, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.09545363485813141, "rewards/total_composite/std": 0.17621159553527832, "reward": 0.09545363485813141, "reward_std": 0.17621161043643951, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09075994789600372, "sampling/sampling_logp_difference/max": 2.8358354568481445, "sampling/importance_sampling_ratio/min": 0.05866949260234833, "sampling/importance_sampling_ratio/mean": 0.9893724918365479, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3145434930920601, "clip_ratio/low_mean": 0.0624148678034544, "clip_ratio/low_min": 0.0624148678034544, "clip_ratio/high_mean": 0.026904894039034843, "clip_ratio/high_max": 0.026904894039034843, "clip_ratio/region_mean": 0.08931976184248924, "reward_total_mean": 0.09545363485813141, "reward_meter_mean": 0.09545363485813141, "reward_meter_std": 0.17621159553527832, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.09545363485813141, "reward_total_composite_std": 0.17621159553527832} {"timestamp_utc": "2026-04-12T01:45:48Z", "mode": "train", "global_step": 2272, "epoch": 0.09125597461541551, "loss": -0.0059, "grad_norm": 5.84548282623291, "learning_rate": 3.1181818181818186e-06, "num_tokens": 5139832.0, "completions/mean_length": 69.0, "completions/min_length": 67.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9690738916397095, "rewards/meter/std": 0.02259281650185585, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9690738916397095, "rewards/total_composite/std": 0.02259281650185585, "reward": 0.9690738916397095, "reward_std": 0.022592833265662193, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031493693590164185, "sampling/sampling_logp_difference/max": 1.107264757156372, "sampling/importance_sampling_ratio/min": 0.33046165108680725, "sampling/importance_sampling_ratio/mean": 1.0030878782272339, "sampling/importance_sampling_ratio/max": 1.7162621021270752, "entropy": 0.21663649939000607, "clip_ratio/low_mean": 0.00911509501747787, "clip_ratio/low_min": 0.00911509501747787, "clip_ratio/high_mean": 0.01436411531176418, "clip_ratio/high_max": 0.01436411531176418, "clip_ratio/region_mean": 0.02347921032924205, "reward_total_mean": 0.9690738916397095, "reward_meter_mean": 0.9690738916397095, "reward_meter_std": 0.02259281650185585, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9690738916397095, "reward_total_composite_std": 0.02259281650185585} {"timestamp_utc": "2026-04-12T01:45:56Z", "mode": "train", "global_step": 2273, "epoch": 0.09129614009720047, "loss": 0.0065, "grad_norm": 2.4946935176849365, "learning_rate": 3.1151515151515154e-06, "num_tokens": 5144049.0, "completions/mean_length": 308.125, "completions/min_length": 295.0, "completions/max_length": 318.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 308.125, "completions/min_terminated_length": 295.0, "completions/max_terminated_length": 318.0, "rewards/meter/mean": 0.985474705696106, "rewards/meter/std": 0.02723194845020771, "rewards/count_adherence/mean": 0.6634615659713745, "rewards/count_adherence/std": 0.039811473339796066, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7977941036224365, "rewards/repeat_penalty/std": 0.08265344798564911, "rewards/total_composite/mean": 0.5205361843109131, "rewards/total_composite/std": 0.053465962409973145, "reward": 0.5205361843109131, "reward_std": 0.05346594378352165, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02342665195465088, "sampling/sampling_logp_difference/max": 1.336355209350586, "sampling/importance_sampling_ratio/min": 0.26280176639556885, "sampling/importance_sampling_ratio/mean": 1.0016851425170898, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.16073014307767153, "clip_ratio/low_mean": 0.011750375153496861, "clip_ratio/low_min": 0.011750375153496861, "clip_ratio/high_mean": 0.008914577425457537, "clip_ratio/high_max": 0.008914577425457537, "clip_ratio/region_mean": 0.0206649525789544, "reward_total_mean": 0.5205361843109131, "reward_meter_mean": 0.985474705696106, "reward_meter_std": 0.02723194845020771, "reward_count_adherence_mean": 0.6634615659713745, "reward_count_adherence_std": 0.039811473339796066, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7977941036224365, "reward_repeat_penalty_std": 0.08265344798564911, "reward_total_composite_mean": 0.5205361843109131, "reward_total_composite_std": 0.053465962409973145} {"timestamp_utc": "2026-04-12T01:46:02Z", "mode": "train", "global_step": 2274, "epoch": 0.09133630557898542, "loss": -0.006, "grad_norm": 2.2683939933776855, "learning_rate": 3.1121212121212126e-06, "num_tokens": 5146825.0, "completions/mean_length": 154.0, "completions/min_length": 150.0, "completions/max_length": 158.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 154.0, "completions/min_terminated_length": 150.0, "completions/max_terminated_length": 158.0, "rewards/meter/mean": 0.9939357042312622, "rewards/meter/std": 0.003742816159501672, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.976262092590332, "rewards/total_composite/std": 0.051798708736896515, "reward": 0.976262092590332, "reward_std": 0.05179871618747711, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03507037088274956, "sampling/sampling_logp_difference/max": 1.2760963439941406, "sampling/importance_sampling_ratio/min": 0.27912476658821106, "sampling/importance_sampling_ratio/mean": 1.006531834602356, "sampling/importance_sampling_ratio/max": 1.880401372909546, "entropy": 0.29750954918563366, "clip_ratio/low_mean": 0.0016666667070239782, "clip_ratio/low_min": 0.0016666667070239782, "clip_ratio/high_mean": 0.02496329549467191, "clip_ratio/high_max": 0.02496329549467191, "clip_ratio/region_mean": 0.02662996220169589, "reward_total_mean": 0.976262092590332, "reward_meter_mean": 0.9939357042312622, "reward_meter_std": 0.003742816159501672, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.976262092590332, "reward_total_composite_std": 0.051798708736896515} {"timestamp_utc": "2026-04-12T01:46:06Z", "mode": "train", "global_step": 2275, "epoch": 0.09137647106077038, "loss": 0.0025, "grad_norm": 2.3654937744140625, "learning_rate": 3.1090909090909095e-06, "num_tokens": 5148537.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.7876189947128296, "rewards/meter/std": 3.669682701001875e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7876189947128296, "rewards/total_composite/std": 3.669682701001875e-05, "reward": 0.7876189947128296, "reward_std": 3.669681245810352e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0021758992224931717, "sampling/sampling_logp_difference/max": 0.25645923614501953, "sampling/importance_sampling_ratio/min": 0.9631562829017639, "sampling/importance_sampling_ratio/mean": 1.0020716190338135, "sampling/importance_sampling_ratio/max": 1.2923461198806763, "entropy": 0.012424270971678197, "clip_ratio/low_mean": 0.002314814832061529, "clip_ratio/low_min": 0.002314814832061529, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.002314814832061529, "reward_total_mean": 0.7876189947128296, "reward_meter_mean": 0.7876189947128296, "reward_meter_std": 3.669682701001875e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7876189947128296, "reward_total_composite_std": 3.669682701001875e-05} {"timestamp_utc": "2026-04-12T01:46:12Z", "mode": "train", "global_step": 2276, "epoch": 0.09141663654255533, "loss": -0.0014, "grad_norm": 1.912965178489685, "learning_rate": 3.1060606060606063e-06, "num_tokens": 5151308.0, "completions/mean_length": 142.375, "completions/min_length": 141.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 142.375, "completions/min_terminated_length": 141.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.9956492185592651, "rewards/meter/std": 0.00017072130867745727, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8611111640930176, "rewards/repeat_penalty/std": 0.05143444612622261, "rewards/total_composite/mean": 0.8573694229125977, "rewards/total_composite/std": 0.05129777640104294, "reward": 0.8573694229125977, "reward_std": 0.05129777267575264, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01284986175596714, "sampling/sampling_logp_difference/max": 1.0029168128967285, "sampling/importance_sampling_ratio/min": 0.3668079972267151, "sampling/importance_sampling_ratio/mean": 1.000401258468628, "sampling/importance_sampling_ratio/max": 1.4935818910598755, "entropy": 0.07997249066829681, "clip_ratio/low_mean": 0.0008865247946232557, "clip_ratio/low_min": 0.0008865247946232557, "clip_ratio/high_mean": 0.009610342094674706, "clip_ratio/high_max": 0.009610342094674706, "clip_ratio/region_mean": 0.010496866889297962, "reward_total_mean": 0.8573694229125977, "reward_meter_mean": 0.9956492185592651, "reward_meter_std": 0.00017072130867745727, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8611111640930176, "reward_repeat_penalty_std": 0.05143444612622261, "reward_total_composite_mean": 0.8573694229125977, "reward_total_composite_std": 0.05129777640104294} {"timestamp_utc": "2026-04-12T01:46:18Z", "mode": "train", "global_step": 2277, "epoch": 0.09145680202434028, "loss": -0.0004, "grad_norm": 2.30692195892334, "learning_rate": 3.103030303030303e-06, "num_tokens": 5154060.0, "completions/mean_length": 170.0, "completions/min_length": 157.0, "completions/max_length": 205.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 170.0, "completions/min_terminated_length": 157.0, "completions/max_terminated_length": 205.0, "rewards/meter/mean": 0.9647563695907593, "rewards/meter/std": 0.03250608593225479, "rewards/count_adherence/mean": 0.8541666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7992424368858337, "rewards/repeat_penalty/std": 0.11809822916984558, "rewards/total_composite/mean": 0.658399760723114, "rewards/total_composite/std": 0.10351461172103882, "reward": 0.658399760723114, "reward_std": 0.10351462662220001, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02872909978032112, "sampling/sampling_logp_difference/max": 1.5842926502227783, "sampling/importance_sampling_ratio/min": 0.20509281754493713, "sampling/importance_sampling_ratio/mean": 0.9995459318161011, "sampling/importance_sampling_ratio/max": 1.7825084924697876, "entropy": 0.18850434944033623, "clip_ratio/low_mean": 0.009245004854165018, "clip_ratio/low_min": 0.009245004854165018, "clip_ratio/high_mean": 0.01534117921255529, "clip_ratio/high_max": 0.01534117921255529, "clip_ratio/region_mean": 0.024586184066720307, "reward_total_mean": 0.658399760723114, "reward_meter_mean": 0.9647563695907593, "reward_meter_std": 0.03250608593225479, "reward_count_adherence_mean": 0.8541666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7992424368858337, "reward_repeat_penalty_std": 0.11809822916984558, "reward_total_composite_mean": 0.658399760723114, "reward_total_composite_std": 0.10351461172103882} {"timestamp_utc": "2026-04-12T01:46:23Z", "mode": "train", "global_step": 2278, "epoch": 0.09149696750612524, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.1000000000000004e-06, "num_tokens": 5155988.0, "completions/mean_length": 80.0, "completions/min_length": 80.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.0, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "reward": 0.7190229296684265, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.000632537470664829, "sampling/sampling_logp_difference/max": 0.022958219051361084, "sampling/importance_sampling_ratio/min": 0.9780369400978088, "sampling/importance_sampling_ratio/mean": 1.0005130767822266, "sampling/importance_sampling_ratio/max": 1.023223876953125, "entropy": 0.006176002731081098, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7190229296684265, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:46:27Z", "mode": "train", "global_step": 2279, "epoch": 0.09153713298791019, "loss": 0.0082, "grad_norm": 13.398720741271973, "learning_rate": 3.0969696969696972e-06, "num_tokens": 5157395.0, "completions/mean_length": 35.875, "completions/min_length": 35.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.875, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9567666053771973, "rewards/meter/std": 0.09672017395496368, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9567666053771973, "rewards/total_composite/std": 0.09672017395496368, "reward": 0.9567666053771973, "reward_std": 0.09672017395496368, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.055583853274583817, "sampling/sampling_logp_difference/max": 1.244422435760498, "sampling/importance_sampling_ratio/min": 0.2881072461605072, "sampling/importance_sampling_ratio/mean": 0.9919366836547852, "sampling/importance_sampling_ratio/max": 1.522891640663147, "entropy": 0.3067398630082607, "clip_ratio/low_mean": 0.0034722222480922937, "clip_ratio/low_min": 0.0034722222480922937, "clip_ratio/high_mean": 0.04533730214461684, "clip_ratio/high_max": 0.04533730214461684, "clip_ratio/region_mean": 0.048809524392709136, "reward_total_mean": 0.9567666053771973, "reward_meter_mean": 0.9567666053771973, "reward_meter_std": 0.09672017395496368, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9567666053771973, "reward_total_composite_std": 0.09672017395496368} {"timestamp_utc": "2026-04-12T01:46:33Z", "mode": "train", "global_step": 2280, "epoch": 0.09157729846969515, "loss": 0.0149, "grad_norm": 5.9251813888549805, "learning_rate": 3.093939393939394e-06, "num_tokens": 5159888.0, "completions/mean_length": 137.625, "completions/min_length": 131.0, "completions/max_length": 143.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 137.625, "completions/min_terminated_length": 131.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.9712404012680054, "rewards/meter/std": 0.020027272403240204, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8324918150901794, "rewards/total_composite/std": 0.017166240140795708, "reward": 0.8324918150901794, "reward_std": 0.017166240140795708, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03175117075443268, "sampling/sampling_logp_difference/max": 1.4596366882324219, "sampling/importance_sampling_ratio/min": 0.2323206663131714, "sampling/importance_sampling_ratio/mean": 1.007045865058899, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.18117596302181482, "clip_ratio/low_mean": 0.009806309011764824, "clip_ratio/low_min": 0.009806309011764824, "clip_ratio/high_mean": 0.008134586038067937, "clip_ratio/high_max": 0.008134586038067937, "clip_ratio/region_mean": 0.01794089504983276, "reward_total_mean": 0.8324918150901794, "reward_meter_mean": 0.9712404012680054, "reward_meter_std": 0.020027272403240204, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8324918150901794, "reward_total_composite_std": 0.017166240140795708} {"timestamp_utc": "2026-04-12T01:46:38Z", "mode": "train", "global_step": 2281, "epoch": 0.0916174639514801, "loss": 0.0073, "grad_norm": 2.3620288372039795, "learning_rate": 3.090909090909091e-06, "num_tokens": 5161346.0, "completions/mean_length": 32.25, "completions/min_length": 32.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.25, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9992298483848572, "rewards/meter/std": 7.94332881923765e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992298483848572, "rewards/total_composite/std": 7.94332881923765e-05, "reward": 0.9992298483848572, "reward_std": 7.940443902043626e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016870969906449318, "sampling/sampling_logp_difference/max": 0.7187650203704834, "sampling/importance_sampling_ratio/min": 0.5924263596534729, "sampling/importance_sampling_ratio/mean": 1.0008995532989502, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.057134561240673065, "clip_ratio/low_mean": 0.007575757801532745, "clip_ratio/low_min": 0.007575757801532745, "clip_ratio/high_mean": 0.015625, "clip_ratio/high_max": 0.015625, "clip_ratio/region_mean": 0.023200757801532745, "reward_total_mean": 0.9992298483848572, "reward_meter_mean": 0.9992298483848572, "reward_meter_std": 7.94332881923765e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992298483848572, "reward_total_composite_std": 7.94332881923765e-05} {"timestamp_utc": "2026-04-12T01:46:43Z", "mode": "train", "global_step": 2282, "epoch": 0.09165762943326505, "loss": 0.0173, "grad_norm": 2.3963494300842285, "learning_rate": 3.087878787878788e-06, "num_tokens": 5163820.0, "completions/mean_length": 132.25, "completions/min_length": 127.0, "completions/max_length": 137.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.25, "completions/min_terminated_length": 127.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.9742553234100342, "rewards/meter/std": 0.023107483983039856, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.1322600096464157, "rewards/total_composite/mean": 0.8361626863479614, "rewards/total_composite/std": 0.13729579746723175, "reward": 0.8361626863479614, "reward_std": 0.13729579746723175, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03228021413087845, "sampling/sampling_logp_difference/max": 1.481900691986084, "sampling/importance_sampling_ratio/min": 0.22720544040203094, "sampling/importance_sampling_ratio/mean": 1.0067442655563354, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23136323876678944, "clip_ratio/low_mean": 0.005548747314605862, "clip_ratio/low_min": 0.005548747314605862, "clip_ratio/high_mean": 0.01910717412829399, "clip_ratio/high_max": 0.01910717412829399, "clip_ratio/region_mean": 0.024655921442899853, "reward_total_mean": 0.8361626863479614, "reward_meter_mean": 0.9742553234100342, "reward_meter_std": 0.023107483983039856, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.1322600096464157, "reward_total_composite_mean": 0.8361626863479614, "reward_total_composite_std": 0.13729579746723175} {"timestamp_utc": "2026-04-12T01:46:49Z", "mode": "train", "global_step": 2283, "epoch": 0.09169779491505001, "loss": 0.0041, "grad_norm": 2.038731098175049, "learning_rate": 3.084848484848485e-06, "num_tokens": 5166635.0, "completions/mean_length": 168.875, "completions/min_length": 167.0, "completions/max_length": 171.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 168.875, "completions/min_terminated_length": 167.0, "completions/max_terminated_length": 171.0, "rewards/meter/mean": 0.9957842230796814, "rewards/meter/std": 0.0002837553038261831, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8636363744735718, "rewards/repeat_penalty/std": 0.08416546881198883, "rewards/total_composite/mean": 0.8599779009819031, "rewards/total_composite/std": 0.08360190689563751, "reward": 0.8599779009819031, "reward_std": 0.08360189944505692, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021027982234954834, "sampling/sampling_logp_difference/max": 1.4786853790283203, "sampling/importance_sampling_ratio/min": 0.22793714702129364, "sampling/importance_sampling_ratio/mean": 1.0022042989730835, "sampling/importance_sampling_ratio/max": 1.8878965377807617, "entropy": 0.11252321023494005, "clip_ratio/low_mean": 0.008858592598699033, "clip_ratio/low_min": 0.008858592598699033, "clip_ratio/high_mean": 0.012570029124617577, "clip_ratio/high_max": 0.012570029124617577, "clip_ratio/region_mean": 0.02142862172331661, "reward_total_mean": 0.8599779009819031, "reward_meter_mean": 0.9957842230796814, "reward_meter_std": 0.0002837553038261831, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8636363744735718, "reward_repeat_penalty_std": 0.08416546881198883, "reward_total_composite_mean": 0.8599779009819031, "reward_total_composite_std": 0.08360190689563751} {"timestamp_utc": "2026-04-12T01:46:53Z", "mode": "train", "global_step": 2284, "epoch": 0.09173796039683496, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.081818181818182e-06, "num_tokens": 5168059.0, "completions/mean_length": 29.0, "completions/min_length": 29.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9929623007774353, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9929623007774353, "rewards/total_composite/std": 0.0, "reward": 0.9929623007774353, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002442288678139448, "sampling/sampling_logp_difference/max": 0.0076376767829060555, "sampling/importance_sampling_ratio/min": 0.9986435174942017, "sampling/importance_sampling_ratio/mean": 1.0002309083938599, "sampling/importance_sampling_ratio/max": 1.0076669454574585, "entropy": 0.0025252392806578428, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9929623007774353, "reward_meter_mean": 0.9929623007774353, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9929623007774353, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:46:59Z", "mode": "train", "global_step": 2285, "epoch": 0.09177812587861992, "loss": -0.0656, "grad_norm": 1.7548184394836426, "learning_rate": 3.078787878787879e-06, "num_tokens": 5170897.0, "completions/mean_length": 171.75, "completions/min_length": 143.0, "completions/max_length": 183.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 171.75, "completions/min_terminated_length": 143.0, "completions/max_terminated_length": 183.0, "rewards/meter/mean": 0.9989954233169556, "rewards/meter/std": 0.0003574988222680986, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9444444179534912, "rewards/repeat_penalty/std": 0.059391383081674576, "rewards/total_composite/mean": 0.8935445547103882, "rewards/total_composite/std": 0.07573916018009186, "reward": 0.8935445547103882, "reward_std": 0.07573915272951126, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022320233285427094, "sampling/sampling_logp_difference/max": 1.9766596555709839, "sampling/importance_sampling_ratio/min": 0.13853120803833008, "sampling/importance_sampling_ratio/mean": 1.0065274238586426, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.19338966719806194, "clip_ratio/low_mean": 0.014906731550581753, "clip_ratio/low_min": 0.014906731550581753, "clip_ratio/high_mean": 0.006868339143693447, "clip_ratio/high_max": 0.006868339143693447, "clip_ratio/region_mean": 0.0217750706942752, "reward_total_mean": 0.8935445547103882, "reward_meter_mean": 0.9989954233169556, "reward_meter_std": 0.0003574988222680986, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9444444179534912, "reward_repeat_penalty_std": 0.059391383081674576, "reward_total_composite_mean": 0.8935445547103882, "reward_total_composite_std": 0.07573916018009186} {"timestamp_utc": "2026-04-12T01:47:04Z", "mode": "train", "global_step": 2286, "epoch": 0.09181829136040487, "loss": -0.0008, "grad_norm": 1.9542547464370728, "learning_rate": 3.075757575757576e-06, "num_tokens": 5173297.0, "completions/mean_length": 106.0, "completions/min_length": 106.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.0, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.680180549621582, "rewards/meter/std": 0.013479826971888542, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5830118656158447, "rewards/total_composite/std": 0.011554129421710968, "reward": 0.5830118656158447, "reward_std": 0.011554128490388393, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0026654843240976334, "sampling/sampling_logp_difference/max": 0.7275528907775879, "sampling/importance_sampling_ratio/min": 0.48308974504470825, "sampling/importance_sampling_ratio/mean": 1.0005828142166138, "sampling/importance_sampling_ratio/max": 1.85325026512146, "entropy": 0.009184476220980287, "clip_ratio/low_mean": 0.002358490601181984, "clip_ratio/low_min": 0.002358490601181984, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.002358490601181984, "reward_total_mean": 0.5830118656158447, "reward_meter_mean": 0.680180549621582, "reward_meter_std": 0.013479826971888542, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5830118656158447, "reward_total_composite_std": 0.011554129421710968} {"timestamp_utc": "2026-04-12T01:47:09Z", "mode": "train", "global_step": 2287, "epoch": 0.09185845684218982, "loss": -0.0098, "grad_norm": 5.288448810577393, "learning_rate": 3.0727272727272727e-06, "num_tokens": 5175104.0, "completions/mean_length": 73.875, "completions/min_length": 70.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.875, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.8793976306915283, "rewards/meter/std": 0.33859992027282715, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8793976306915283, "rewards/total_composite/std": 0.33859992027282715, "reward": 0.8793976306915283, "reward_std": 0.33859992027282715, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027358917519450188, "sampling/sampling_logp_difference/max": 0.9376602172851562, "sampling/importance_sampling_ratio/min": 0.39154288172721863, "sampling/importance_sampling_ratio/mean": 1.005669355392456, "sampling/importance_sampling_ratio/max": 1.7006129026412964, "entropy": 0.24043426476418972, "clip_ratio/low_mean": 0.0017857142956927419, "clip_ratio/low_min": 0.0017857142956927419, "clip_ratio/high_mean": 0.016635622712783515, "clip_ratio/high_max": 0.016635622712783515, "clip_ratio/region_mean": 0.018421337008476257, "reward_total_mean": 0.8793976306915283, "reward_meter_mean": 0.8793976306915283, "reward_meter_std": 0.33859992027282715, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8793976306915283, "reward_total_composite_std": 0.33859992027282715} {"timestamp_utc": "2026-04-12T01:47:15Z", "mode": "train", "global_step": 2288, "epoch": 0.09189862232397478, "loss": 0.013, "grad_norm": 7.310388088226318, "learning_rate": 3.0696969696969696e-06, "num_tokens": 5178040.0, "completions/mean_length": 168.0, "completions/min_length": 166.0, "completions/max_length": 170.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 168.0, "completions/min_terminated_length": 166.0, "completions/max_terminated_length": 170.0, "rewards/meter/mean": 0.7916620969772339, "rewards/meter/std": 0.292292982339859, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.930555522441864, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.621074378490448, "rewards/total_composite/std": 0.2475537359714508, "reward": 0.621074378490448, "reward_std": 0.2475537210702896, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03382183238863945, "sampling/sampling_logp_difference/max": 1.4603016376495361, "sampling/importance_sampling_ratio/min": 0.23216624557971954, "sampling/importance_sampling_ratio/mean": 1.000993251800537, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17776653543114662, "clip_ratio/low_mean": 0.008171421475708485, "clip_ratio/low_min": 0.008171421475708485, "clip_ratio/high_mean": 0.017823605565354228, "clip_ratio/high_max": 0.017823605565354228, "clip_ratio/region_mean": 0.025995027041062713, "reward_total_mean": 0.621074378490448, "reward_meter_mean": 0.7916620969772339, "reward_meter_std": 0.292292982339859, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.930555522441864, "reward_repeat_penalty_std": 0.05750546231865883, "reward_total_composite_mean": 0.621074378490448, "reward_total_composite_std": 0.2475537359714508} {"timestamp_utc": "2026-04-12T01:47:19Z", "mode": "train", "global_step": 2289, "epoch": 0.09193878780575973, "loss": 0.012, "grad_norm": 1.6638261079788208, "learning_rate": 3.066666666666667e-06, "num_tokens": 5179986.0, "completions/mean_length": 68.25, "completions/min_length": 66.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.25, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9976264834403992, "rewards/meter/std": 0.00037816373514942825, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9976264834403992, "rewards/total_composite/std": 0.00037816373514942825, "reward": 0.9976264834403992, "reward_std": 0.00037817005068063736, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01430835947394371, "sampling/sampling_logp_difference/max": 2.2254679203033447, "sampling/importance_sampling_ratio/min": 0.10801687091588974, "sampling/importance_sampling_ratio/mean": 0.996768593788147, "sampling/importance_sampling_ratio/max": 1.2755680084228516, "entropy": 0.04147487785667181, "clip_ratio/low_mean": 0.005434782709926367, "clip_ratio/low_min": 0.005434782709926367, "clip_ratio/high_mean": 0.005681818351149559, "clip_ratio/high_max": 0.005681818351149559, "clip_ratio/region_mean": 0.011116601061075926, "reward_total_mean": 0.9976264834403992, "reward_meter_mean": 0.9976264834403992, "reward_meter_std": 0.00037816373514942825, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9976264834403992, "reward_total_composite_std": 0.00037816373514942825} {"timestamp_utc": "2026-04-12T01:47:24Z", "mode": "train", "global_step": 2290, "epoch": 0.09197895328754468, "loss": 0.0052, "grad_norm": 1.542847752571106, "learning_rate": 3.0636363636363636e-06, "num_tokens": 5181793.0, "completions/mean_length": 64.875, "completions/min_length": 64.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.875, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9991232752799988, "rewards/meter/std": 0.00012234247697051615, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991232752799988, "rewards/total_composite/std": 0.00012234247697051615, "reward": 0.9991232752799988, "reward_std": 0.00012233950837980956, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01243247464299202, "sampling/sampling_logp_difference/max": 1.3001289367675781, "sampling/importance_sampling_ratio/min": 0.27249664068222046, "sampling/importance_sampling_ratio/mean": 0.9971635937690735, "sampling/importance_sampling_ratio/max": 1.3870518207550049, "entropy": 0.04461120581254363, "clip_ratio/low_mean": 0.001923076924867928, "clip_ratio/low_min": 0.001923076924867928, "clip_ratio/high_mean": 0.005859375, "clip_ratio/high_max": 0.005859375, "clip_ratio/region_mean": 0.007782451924867928, "reward_total_mean": 0.9991232752799988, "reward_meter_mean": 0.9991232752799988, "reward_meter_std": 0.00012234247697051615, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9991232752799988, "reward_total_composite_std": 0.00012234247697051615} {"timestamp_utc": "2026-04-12T01:47:28Z", "mode": "train", "global_step": 2291, "epoch": 0.09201911876932964, "loss": -0.0007, "grad_norm": 0.26676464080810547, "learning_rate": 3.0606060606060605e-06, "num_tokens": 5183553.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.998096764087677, "rewards/meter/std": 3.926327553926967e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998096764087677, "rewards/total_composite/std": 3.926327553926967e-05, "reward": 0.998096764087677, "reward_std": 3.925973942386918e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005647474434226751, "sampling/sampling_logp_difference/max": 0.8950796127319336, "sampling/importance_sampling_ratio/min": 0.4085750877857208, "sampling/importance_sampling_ratio/mean": 0.999199628829956, "sampling/importance_sampling_ratio/max": 1.0820250511169434, "entropy": 0.036194659769535065, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0018939394503831863, "clip_ratio/high_max": 0.0018939394503831863, "clip_ratio/region_mean": 0.0018939394503831863, "reward_total_mean": 0.998096764087677, "reward_meter_mean": 0.998096764087677, "reward_meter_std": 3.926327553926967e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.998096764087677, "reward_total_composite_std": 3.926327553926967e-05} {"timestamp_utc": "2026-04-12T01:47:34Z", "mode": "train", "global_step": 2292, "epoch": 0.09205928425111459, "loss": -0.0461, "grad_norm": 3.2664194107055664, "learning_rate": 3.057575757575758e-06, "num_tokens": 5186427.0, "completions/mean_length": 159.25, "completions/min_length": 148.0, "completions/max_length": 201.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 159.25, "completions/min_terminated_length": 148.0, "completions/max_terminated_length": 201.0, "rewards/meter/mean": 0.9959465861320496, "rewards/meter/std": 0.003204776206985116, "rewards/count_adherence/mean": 0.8250000476837158, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9682539701461792, "rewards/repeat_penalty/std": 0.059391383081674576, "rewards/total_composite/mean": 0.793583869934082, "rewards/total_composite/std": 0.05434796214103699, "reward": 0.793583869934082, "reward_std": 0.05434795841574669, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03989114239811897, "sampling/sampling_logp_difference/max": 2.4048454761505127, "sampling/importance_sampling_ratio/min": 0.09027944505214691, "sampling/importance_sampling_ratio/mean": 1.0047212839126587, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.30428963899612427, "clip_ratio/low_mean": 0.005605590064078569, "clip_ratio/low_min": 0.005605590064078569, "clip_ratio/high_mean": 0.024386633071117103, "clip_ratio/high_max": 0.024386633071117103, "clip_ratio/region_mean": 0.029992223135195673, "reward_total_mean": 0.793583869934082, "reward_meter_mean": 0.9959465861320496, "reward_meter_std": 0.003204776206985116, "reward_count_adherence_mean": 0.8250000476837158, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9682539701461792, "reward_repeat_penalty_std": 0.059391383081674576, "reward_total_composite_mean": 0.793583869934082, "reward_total_composite_std": 0.05434796214103699} {"timestamp_utc": "2026-04-12T01:47:42Z", "mode": "train", "global_step": 2293, "epoch": 0.09209944973289955, "loss": 0.0327, "grad_norm": 1.504547119140625, "learning_rate": 3.054545454545455e-06, "num_tokens": 5191210.0, "completions/mean_length": 363.875, "completions/min_length": 343.0, "completions/max_length": 392.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 363.875, "completions/min_terminated_length": 343.0, "completions/max_terminated_length": 392.0, "rewards/meter/mean": 0.9957020282745361, "rewards/meter/std": 0.001336081069894135, "rewards/count_adherence/mean": 0.6083333492279053, "rewards/count_adherence/std": 0.0235702246427536, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8192294836044312, "rewards/repeat_penalty/std": 0.07807844877243042, "rewards/total_composite/mean": 0.4955565929412842, "rewards/total_composite/std": 0.04254494607448578, "reward": 0.4955565929412842, "reward_std": 0.04254494979977608, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.032130178064107895, "sampling/sampling_logp_difference/max": 2.1043951511383057, "sampling/importance_sampling_ratio/min": 0.12191939353942871, "sampling/importance_sampling_ratio/mean": 1.0060222148895264, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.26295676827430725, "clip_ratio/low_mean": 0.010282430681400001, "clip_ratio/low_min": 0.010282430681400001, "clip_ratio/high_mean": 0.017152261221781373, "clip_ratio/high_max": 0.017152261221781373, "clip_ratio/region_mean": 0.027434691903181374, "reward_total_mean": 0.4955565929412842, "reward_meter_mean": 0.9957020282745361, "reward_meter_std": 0.001336081069894135, "reward_count_adherence_mean": 0.6083333492279053, "reward_count_adherence_std": 0.0235702246427536, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8192294836044312, "reward_repeat_penalty_std": 0.07807844877243042, "reward_total_composite_mean": 0.4955565929412842, "reward_total_composite_std": 0.04254494607448578} {"timestamp_utc": "2026-04-12T01:47:51Z", "mode": "train", "global_step": 2294, "epoch": 0.0921396152146845, "loss": -0.0027, "grad_norm": 1.8439388275146484, "learning_rate": 3.051515151515152e-06, "num_tokens": 5195522.0, "completions/mean_length": 309.0, "completions/min_length": 287.0, "completions/max_length": 326.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 309.0, "completions/min_terminated_length": 287.0, "completions/max_terminated_length": 326.0, "rewards/meter/mean": 0.99519944190979, "rewards/meter/std": 0.003783464664593339, "rewards/count_adherence/mean": 0.7875000238418579, "rewards/count_adherence/std": 0.0353553481400013, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8807692527770996, "rewards/repeat_penalty/std": 0.07008695602416992, "rewards/total_composite/mean": 0.690649151802063, "rewards/total_composite/std": 0.06761835515499115, "reward": 0.690649151802063, "reward_std": 0.06761834770441055, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03343411162495613, "sampling/sampling_logp_difference/max": 1.6177313327789307, "sampling/importance_sampling_ratio/min": 0.19834816455841064, "sampling/importance_sampling_ratio/mean": 1.0094666481018066, "sampling/importance_sampling_ratio/max": 1.937671184539795, "entropy": 0.25325608998537064, "clip_ratio/low_mean": 0.00620853912550956, "clip_ratio/low_min": 0.00620853912550956, "clip_ratio/high_mean": 0.014976657228544354, "clip_ratio/high_max": 0.014976657228544354, "clip_ratio/region_mean": 0.021185196354053915, "reward_total_mean": 0.690649151802063, "reward_meter_mean": 0.99519944190979, "reward_meter_std": 0.003783464664593339, "reward_count_adherence_mean": 0.7875000238418579, "reward_count_adherence_std": 0.0353553481400013, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8807692527770996, "reward_repeat_penalty_std": 0.07008695602416992, "reward_total_composite_mean": 0.690649151802063, "reward_total_composite_std": 0.06761835515499115} {"timestamp_utc": "2026-04-12T01:47:55Z", "mode": "train", "global_step": 2295, "epoch": 0.09217978069646945, "loss": 0.0008, "grad_norm": 3.1970479488372803, "learning_rate": 3.048484848484849e-06, "num_tokens": 5197722.0, "completions/mean_length": 106.0, "completions/min_length": 106.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.0, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.6495559215545654, "rewards/meter/std": 0.1058802381157875, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5567622184753418, "rewards/total_composite/std": 0.09075448662042618, "reward": 0.5567622184753418, "reward_std": 0.09075447171926498, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003019496565684676, "sampling/sampling_logp_difference/max": 0.5486016273498535, "sampling/importance_sampling_ratio/min": 0.5777571797370911, "sampling/importance_sampling_ratio/mean": 0.9996184706687927, "sampling/importance_sampling_ratio/max": 1.2651506662368774, "entropy": 0.02729532145895064, "clip_ratio/low_mean": 0.001179245300590992, "clip_ratio/low_min": 0.001179245300590992, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.001179245300590992, "reward_total_mean": 0.5567622184753418, "reward_meter_mean": 0.6495559215545654, "reward_meter_std": 0.1058802381157875, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5567622184753418, "reward_total_composite_std": 0.09075448662042618} {"timestamp_utc": "2026-04-12T01:48:00Z", "mode": "train", "global_step": 2296, "epoch": 0.09221994617825441, "loss": 0.005, "grad_norm": 2.241980791091919, "learning_rate": 3.045454545454546e-06, "num_tokens": 5199420.0, "completions/mean_length": 66.25, "completions/min_length": 66.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.25, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.998041033744812, "rewards/meter/std": 8.582558075431734e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998041033744812, "rewards/total_composite/std": 8.582558075431734e-05, "reward": 0.998041033744812, "reward_std": 8.582806185586378e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014799090102314949, "sampling/sampling_logp_difference/max": 1.3824272155761719, "sampling/importance_sampling_ratio/min": 0.25096866488456726, "sampling/importance_sampling_ratio/mean": 0.99764084815979, "sampling/importance_sampling_ratio/max": 1.2699003219604492, "entropy": 0.06576015800237656, "clip_ratio/low_mean": 0.0018382353009656072, "clip_ratio/low_min": 0.0018382353009656072, "clip_ratio/high_mean": 0.0037878789007663727, "clip_ratio/high_max": 0.0037878789007663727, "clip_ratio/region_mean": 0.00562611420173198, "reward_total_mean": 0.998041033744812, "reward_meter_mean": 0.998041033744812, "reward_meter_std": 8.582558075431734e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.998041033744812, "reward_total_composite_std": 8.582558075431734e-05} {"timestamp_utc": "2026-04-12T01:48:04Z", "mode": "train", "global_step": 2297, "epoch": 0.09226011166003936, "loss": 0.0003, "grad_norm": 0.5168666839599609, "learning_rate": 3.0424242424242427e-06, "num_tokens": 5201164.0, "completions/mean_length": 55.0, "completions/min_length": 55.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.0, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9948955774307251, "rewards/meter/std": 1.7123466022894718e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9948955774307251, "rewards/total_composite/std": 1.7123466022894718e-05, "reward": 0.9948955774307251, "reward_std": 1.7113467038143426e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0070167421363294125, "sampling/sampling_logp_difference/max": 0.6335644721984863, "sampling/importance_sampling_ratio/min": 0.5306968092918396, "sampling/importance_sampling_ratio/mean": 0.9979526400566101, "sampling/importance_sampling_ratio/max": 1.3264281749725342, "entropy": 0.037125169299542904, "clip_ratio/low_mean": 0.0022727272007614374, "clip_ratio/low_min": 0.0022727272007614374, "clip_ratio/high_mean": 0.0022727272007614374, "clip_ratio/high_max": 0.0022727272007614374, "clip_ratio/region_mean": 0.004545454401522875, "reward_total_mean": 0.9948955774307251, "reward_meter_mean": 0.9948955774307251, "reward_meter_std": 1.7123466022894718e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9948955774307251, "reward_total_composite_std": 1.7123466022894718e-05} {"timestamp_utc": "2026-04-12T01:48:08Z", "mode": "train", "global_step": 2298, "epoch": 0.09230027714182432, "loss": -0.0044, "grad_norm": 2.7533559799194336, "learning_rate": 3.03939393939394e-06, "num_tokens": 5203008.0, "completions/mean_length": 76.5, "completions/min_length": 75.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.5, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9956209659576416, "rewards/meter/std": 0.0032750838436186314, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9956209659576416, "rewards/total_composite/std": 0.0032750838436186314, "reward": 0.9956209659576416, "reward_std": 0.00327507546171546, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0241053719073534, "sampling/sampling_logp_difference/max": 1.008596420288086, "sampling/importance_sampling_ratio/min": 0.3647305369377136, "sampling/importance_sampling_ratio/mean": 1.0014814138412476, "sampling/importance_sampling_ratio/max": 1.4816969633102417, "entropy": 0.21845361031591892, "clip_ratio/low_mean": 0.004934210330247879, "clip_ratio/low_min": 0.004934210330247879, "clip_ratio/high_mean": 0.014611484948545694, "clip_ratio/high_max": 0.014611484948545694, "clip_ratio/region_mean": 0.019545695278793573, "reward_total_mean": 0.9956209659576416, "reward_meter_mean": 0.9956209659576416, "reward_meter_std": 0.0032750838436186314, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9956209659576416, "reward_total_composite_std": 0.0032750838436186314} {"timestamp_utc": "2026-04-12T01:48:14Z", "mode": "train", "global_step": 2299, "epoch": 0.09234044262360927, "loss": 0.013, "grad_norm": 4.413851737976074, "learning_rate": 3.036363636363637e-06, "num_tokens": 5205419.0, "completions/mean_length": 136.375, "completions/min_length": 131.0, "completions/max_length": 143.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 136.375, "completions/min_terminated_length": 131.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.8257510662078857, "rewards/meter/std": 0.3110801577568054, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9285714626312256, "rewards/repeat_penalty/std": 0.1079898476600647, "rewards/total_composite/mean": 0.754673421382904, "rewards/total_composite/std": 0.2842678129673004, "reward": 0.754673421382904, "reward_std": 0.284267783164978, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02921801060438156, "sampling/sampling_logp_difference/max": 1.611825942993164, "sampling/importance_sampling_ratio/min": 0.1995229572057724, "sampling/importance_sampling_ratio/mean": 1.0080208778381348, "sampling/importance_sampling_ratio/max": 1.8313679695129395, "entropy": 0.24156404845416546, "clip_ratio/low_mean": 0.01099004433490336, "clip_ratio/low_min": 0.01099004433490336, "clip_ratio/high_mean": 0.02088787977118045, "clip_ratio/high_max": 0.02088787977118045, "clip_ratio/region_mean": 0.03187792410608381, "reward_total_mean": 0.754673421382904, "reward_meter_mean": 0.8257510662078857, "reward_meter_std": 0.3110801577568054, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9285714626312256, "reward_repeat_penalty_std": 0.1079898476600647, "reward_total_composite_mean": 0.754673421382904, "reward_total_composite_std": 0.2842678129673004} {"timestamp_utc": "2026-04-12T01:48:18Z", "mode": "train", "global_step": 2300, "epoch": 0.09238060810539422, "loss": -0.0163, "grad_norm": 2.978400230407715, "learning_rate": 3.0333333333333337e-06, "num_tokens": 5207160.0, "completions/mean_length": 65.625, "completions/min_length": 62.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.625, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.996369481086731, "rewards/meter/std": 0.00487491674721241, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.996369481086731, "rewards/total_composite/std": 0.00487491674721241, "reward": 0.996369481086731, "reward_std": 0.004874910227954388, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007210928946733475, "sampling/sampling_logp_difference/max": 0.5058443546295166, "sampling/importance_sampling_ratio/min": 0.7272458076477051, "sampling/importance_sampling_ratio/mean": 1.004469871520996, "sampling/importance_sampling_ratio/max": 1.6583852767944336, "entropy": 0.04499326320365071, "clip_ratio/low_mean": 0.002016128972172737, "clip_ratio/low_min": 0.002016128972172737, "clip_ratio/high_mean": 0.0018939394503831863, "clip_ratio/high_max": 0.0018939394503831863, "clip_ratio/region_mean": 0.0039100684225559235, "reward_total_mean": 0.996369481086731, "reward_meter_mean": 0.996369481086731, "reward_meter_std": 0.00487491674721241, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.996369481086731, "reward_total_composite_std": 0.00487491674721241} {"timestamp_utc": "2026-04-12T01:49:20Z", "mode": "eval", "global_step": 2300, "epoch": 0.09238060810539422, "eval_loss": NaN, "eval_runtime": 61.8872, "eval_samples_per_second": 1.68, "eval_steps_per_second": 0.21, "eval_num_tokens": 5207160.0, "eval_completions/mean_length": 179.41346153846155, "eval_completions/min_length": 60.46153846153846, "eval_completions/max_length": 324.3076923076923, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 179.41346153846155, "eval_completions/min_terminated_length": 60.46153846153846, "eval_completions/max_terminated_length": 324.3076923076923, "eval_rewards/meter/mean": 0.7395852987582867, "eval_rewards/meter/std": 0.3918028657252972, "eval_rewards/count_adherence/mean": 0.8744415503281814, "eval_rewards/count_adherence/std": 0.1353673808849775, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/repeat_penalty/mean": 0.9028620398961581, "eval_rewards/repeat_penalty/std": 0.10280066442031127, "eval_rewards/total_composite/mean": 0.5927730821646177, "eval_rewards/total_composite/std": 0.35657222683613116, "eval_reward": 0.5927730821646177, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.02075868207388199, "eval_sampling/sampling_logp_difference/max": 0.9916172761183518, "eval_sampling/importance_sampling_ratio/min": 0.38055819043746364, "eval_sampling/importance_sampling_ratio/mean": 1.004779577255249, "eval_sampling/importance_sampling_ratio/max": 1.405327586027292, "eval_entropy": 0.21972922522288102, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5927730821646177, "eval_reward_meter_mean": 0.7395852987582867, "eval_reward_meter_std": 0.3918028657252972, "eval_reward_count_adherence_mean": 0.8744415503281814, "eval_reward_count_adherence_std": 0.1353673808849775, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_repeat_penalty_mean": 0.9028620398961581, "eval_reward_repeat_penalty_std": 0.10280066442031127, "eval_reward_total_composite_mean": 0.5927730821646177, "eval_reward_total_composite_std": 0.35657222683613116} {"timestamp_utc": "2026-04-12T01:49:27Z", "mode": "train", "global_step": 2301, "epoch": 0.09242077358717918, "loss": -0.0022, "grad_norm": 1.7726956605911255, "learning_rate": 3.0303030303030305e-06, "num_tokens": 5209084.0, "completions/mean_length": 73.5, "completions/min_length": 73.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.5, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9988920092582703, "rewards/meter/std": 0.0004005288355983794, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9988920092582703, "rewards/total_composite/std": 0.0004005288355983794, "reward": 0.9988920092582703, "reward_std": 0.0004005288355983794, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023180123418569565, "sampling/sampling_logp_difference/max": 0.5834887027740479, "sampling/importance_sampling_ratio/min": 0.5579484701156616, "sampling/importance_sampling_ratio/mean": 1.0048654079437256, "sampling/importance_sampling_ratio/max": 1.46310555934906, "entropy": 0.21760859712958336, "clip_ratio/low_mean": 0.008561643771827221, "clip_ratio/low_min": 0.008561643771827221, "clip_ratio/high_mean": 0.01874305820092559, "clip_ratio/high_max": 0.01874305820092559, "clip_ratio/region_mean": 0.02730470197275281, "reward_total_mean": 0.9988920092582703, "reward_meter_mean": 0.9988920092582703, "reward_meter_std": 0.0004005288355983794, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9988920092582703, "reward_total_composite_std": 0.0004005288355983794} {"timestamp_utc": "2026-04-12T01:49:32Z", "mode": "train", "global_step": 2302, "epoch": 0.09246093906896413, "loss": 0.0094, "grad_norm": 4.9054951667785645, "learning_rate": 3.0272727272727277e-06, "num_tokens": 5210937.0, "completions/mean_length": 68.625, "completions/min_length": 67.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.625, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.8321632742881775, "rewards/meter/std": 0.27767401933670044, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8321632742881775, "rewards/total_composite/std": 0.27767401933670044, "reward": 0.8321632742881775, "reward_std": 0.27767398953437805, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03458629176020622, "sampling/sampling_logp_difference/max": 1.4874401092529297, "sampling/importance_sampling_ratio/min": 0.22595034539699554, "sampling/importance_sampling_ratio/mean": 1.0079941749572754, "sampling/importance_sampling_ratio/max": 1.5462456941604614, "entropy": 0.26816077157855034, "clip_ratio/low_mean": 0.0018115942366421223, "clip_ratio/low_min": 0.0018115942366421223, "clip_ratio/high_mean": 0.025409106630831957, "clip_ratio/high_max": 0.025409106630831957, "clip_ratio/region_mean": 0.02722070086747408, "reward_total_mean": 0.8321632742881775, "reward_meter_mean": 0.8321632742881775, "reward_meter_std": 0.27767401933670044, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8321632742881775, "reward_total_composite_std": 0.27767401933670044} {"timestamp_utc": "2026-04-12T01:49:41Z", "mode": "train", "global_step": 2303, "epoch": 0.09250110455074909, "loss": -0.0043, "grad_norm": 1.882811188697815, "learning_rate": 3.0242424242424246e-06, "num_tokens": 5215896.0, "completions/mean_length": 387.875, "completions/min_length": 366.0, "completions/max_length": 422.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 387.875, "completions/min_terminated_length": 366.0, "completions/max_terminated_length": 422.0, "rewards/meter/mean": 0.9971157908439636, "rewards/meter/std": 0.001221620594151318, "rewards/count_adherence/mean": 0.5882353186607361, "rewards/count_adherence/std": 0.03144249692559242, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9170373678207397, "rewards/repeat_penalty/std": 0.06482076644897461, "rewards/total_composite/mean": 0.5379288196563721, "rewards/total_composite/std": 0.04898425191640854, "reward": 0.5379288196563721, "reward_std": 0.04898424819111824, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.035993542522192, "sampling/sampling_logp_difference/max": 2.7829220294952393, "sampling/importance_sampling_ratio/min": 0.25277209281921387, "sampling/importance_sampling_ratio/mean": 1.0075607299804688, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2818215675652027, "clip_ratio/low_mean": 0.010549622820690274, "clip_ratio/low_min": 0.010549622820690274, "clip_ratio/high_mean": 0.011514384881593287, "clip_ratio/high_max": 0.011514384881593287, "clip_ratio/region_mean": 0.02206400770228356, "reward_total_mean": 0.5379288196563721, "reward_meter_mean": 0.9971157908439636, "reward_meter_std": 0.001221620594151318, "reward_count_adherence_mean": 0.5882353186607361, "reward_count_adherence_std": 0.03144249692559242, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9170373678207397, "reward_repeat_penalty_std": 0.06482076644897461, "reward_total_composite_mean": 0.5379288196563721, "reward_total_composite_std": 0.04898425191640854} {"timestamp_utc": "2026-04-12T01:49:45Z", "mode": "train", "global_step": 2304, "epoch": 0.09254127003253404, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.0212121212121214e-06, "num_tokens": 5217688.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00046610343270003796, "sampling/sampling_logp_difference/max": 0.01851455122232437, "sampling/importance_sampling_ratio/min": 0.9936742186546326, "sampling/importance_sampling_ratio/mean": 1.000432014465332, "sampling/importance_sampling_ratio/max": 1.0186870098114014, "entropy": 0.004575719474814832, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:49:51Z", "mode": "train", "global_step": 2305, "epoch": 0.092581435514319, "loss": -0.0071, "grad_norm": 2.1037492752075195, "learning_rate": 3.0181818181818182e-06, "num_tokens": 5220865.0, "completions/mean_length": 188.125, "completions/min_length": 184.0, "completions/max_length": 192.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 188.125, "completions/min_terminated_length": 184.0, "completions/max_terminated_length": 192.0, "rewards/meter/mean": 0.9963282942771912, "rewards/meter/std": 0.0020138672553002834, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9305555820465088, "rewards/repeat_penalty/std": 0.08266931027173996, "rewards/total_composite/mean": 0.7725695371627808, "rewards/total_composite/std": 0.06809280812740326, "reward": 0.7725695371627808, "reward_std": 0.06809281557798386, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.033207789063453674, "sampling/sampling_logp_difference/max": 1.2795171737670898, "sampling/importance_sampling_ratio/min": 0.278171569108963, "sampling/importance_sampling_ratio/mean": 1.0070801973342896, "sampling/importance_sampling_ratio/max": 1.9226046800613403, "entropy": 0.2865954786539078, "clip_ratio/low_mean": 0.014036171836778522, "clip_ratio/low_min": 0.014036171836778522, "clip_ratio/high_mean": 0.014565107179805636, "clip_ratio/high_max": 0.014565107179805636, "clip_ratio/region_mean": 0.028601279016584158, "reward_total_mean": 0.7725695371627808, "reward_meter_mean": 0.9963282942771912, "reward_meter_std": 0.0020138672553002834, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9305555820465088, "reward_repeat_penalty_std": 0.08266931027173996, "reward_total_composite_mean": 0.7725695371627808, "reward_total_composite_std": 0.06809280812740326} {"timestamp_utc": "2026-04-12T01:49:55Z", "mode": "train", "global_step": 2306, "epoch": 0.09262160099610395, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.0151515151515155e-06, "num_tokens": 5222265.0, "completions/mean_length": 35.0, "completions/min_length": 35.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.997984766960144, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.997984766960144, "rewards/total_composite/std": 0.0, "reward": 0.997984766960144, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00116938806604594, "sampling/sampling_logp_difference/max": 0.035293105989694595, "sampling/importance_sampling_ratio/min": 0.9653224349021912, "sampling/importance_sampling_ratio/mean": 1.0008283853530884, "sampling/importance_sampling_ratio/max": 1.025589942932129, "entropy": 0.009295969794038683, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.997984766960144, "reward_meter_mean": 0.997984766960144, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.997984766960144, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:50:00Z", "mode": "train", "global_step": 2307, "epoch": 0.0926617664778889, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.0121212121212123e-06, "num_tokens": 5223977.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0005960848648101091, "sampling/sampling_logp_difference/max": 0.03551316633820534, "sampling/importance_sampling_ratio/min": 0.9723794460296631, "sampling/importance_sampling_ratio/mean": 1.0003576278686523, "sampling/importance_sampling_ratio/max": 1.0361512899398804, "entropy": 0.00621713837608695, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:50:04Z", "mode": "train", "global_step": 2308, "epoch": 0.09270193195967386, "loss": -0.0001, "grad_norm": 1.3243682384490967, "learning_rate": 3.009090909090909e-06, "num_tokens": 5225489.0, "completions/mean_length": 32.0, "completions/min_length": 32.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9992917776107788, "rewards/meter/std": 1.824959326768294e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992917776107788, "rewards/total_composite/std": 1.824959326768294e-05, "reward": 0.9992917776107788, "reward_std": 1.82435705937678e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0022324356250464916, "sampling/sampling_logp_difference/max": 0.30353307723999023, "sampling/importance_sampling_ratio/min": 0.7382054924964905, "sampling/importance_sampling_ratio/mean": 0.9991614818572998, "sampling/importance_sampling_ratio/max": 1.0328819751739502, "entropy": 0.007502294494770467, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9992917776107788, "reward_meter_mean": 0.9992917776107788, "reward_meter_std": 1.824959326768294e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992917776107788, "reward_total_composite_std": 1.824959326768294e-05} {"timestamp_utc": "2026-04-12T01:50:08Z", "mode": "train", "global_step": 2309, "epoch": 0.09274209744145881, "loss": 0.0018, "grad_norm": 1.3253436088562012, "learning_rate": 3.0060606060606064e-06, "num_tokens": 5227402.0, "completions/mean_length": 64.125, "completions/min_length": 64.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.125, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9992837905883789, "rewards/meter/std": 7.721914880676195e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992837905883789, "rewards/total_composite/std": 7.721914880676195e-05, "reward": 0.9992837905883789, "reward_std": 7.722593727521598e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004359652288258076, "sampling/sampling_logp_difference/max": 1.3222827911376953, "sampling/importance_sampling_ratio/min": 0.7730492949485779, "sampling/importance_sampling_ratio/mean": 1.002130150794983, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.013516030623577535, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9992837905883789, "reward_meter_mean": 0.9992837905883789, "reward_meter_std": 7.721914880676195e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992837905883789, "reward_total_composite_std": 7.721914880676195e-05} {"timestamp_utc": "2026-04-12T01:50:12Z", "mode": "train", "global_step": 2310, "epoch": 0.09278226292324376, "loss": -0.001, "grad_norm": 0.4977571368217468, "learning_rate": 3.0030303030303032e-06, "num_tokens": 5229178.0, "completions/mean_length": 55.0, "completions/min_length": 55.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.0, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9948618412017822, "rewards/meter/std": 9.404453885508701e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9948618412017822, "rewards/total_composite/std": 9.404453885508701e-05, "reward": 0.9948618412017822, "reward_std": 9.405182208865881e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008051915094256401, "sampling/sampling_logp_difference/max": 0.535893440246582, "sampling/importance_sampling_ratio/min": 0.5851462483406067, "sampling/importance_sampling_ratio/mean": 1.000832200050354, "sampling/importance_sampling_ratio/max": 1.473840355873108, "entropy": 0.03565134946256876, "clip_ratio/low_mean": 0.004545454401522875, "clip_ratio/low_min": 0.004545454401522875, "clip_ratio/high_mean": 0.006818181602284312, "clip_ratio/high_max": 0.006818181602284312, "clip_ratio/region_mean": 0.011363636003807187, "reward_total_mean": 0.9948618412017822, "reward_meter_mean": 0.9948618412017822, "reward_meter_std": 9.404453885508701e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9948618412017822, "reward_total_composite_std": 9.404453885508701e-05} {"timestamp_utc": "2026-04-12T01:50:16Z", "mode": "train", "global_step": 2311, "epoch": 0.09282242840502872, "loss": -0.0002, "grad_norm": 0.3652920722961426, "learning_rate": 3e-06, "num_tokens": 5230866.0, "completions/mean_length": 55.0, "completions/min_length": 55.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.0, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9948963522911072, "rewards/meter/std": 1.816852636693511e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9948963522911072, "rewards/total_composite/std": 1.816852636693511e-05, "reward": 0.9948963522911072, "reward_std": 1.817071097320877e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00472200708463788, "sampling/sampling_logp_difference/max": 0.3054179549217224, "sampling/importance_sampling_ratio/min": 0.7490553259849548, "sampling/importance_sampling_ratio/mean": 0.999363124370575, "sampling/importance_sampling_ratio/max": 1.3571921586990356, "entropy": 0.032598731108009815, "clip_ratio/low_mean": 0.0022727272007614374, "clip_ratio/low_min": 0.0022727272007614374, "clip_ratio/high_mean": 0.004545454401522875, "clip_ratio/high_max": 0.004545454401522875, "clip_ratio/region_mean": 0.006818181602284312, "reward_total_mean": 0.9948963522911072, "reward_meter_mean": 0.9948963522911072, "reward_meter_std": 1.816852636693511e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9948963522911072, "reward_total_composite_std": 1.816852636693511e-05} {"timestamp_utc": "2026-04-12T01:50:21Z", "mode": "train", "global_step": 2312, "epoch": 0.09286259388681367, "loss": -0.0133, "grad_norm": 2.4858875274658203, "learning_rate": 2.996969696969697e-06, "num_tokens": 5232710.0, "completions/mean_length": 76.5, "completions/min_length": 75.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.5, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9963726997375488, "rewards/meter/std": 0.000908972229808569, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9963726997375488, "rewards/total_composite/std": 0.000908972229808569, "reward": 0.9963726997375488, "reward_std": 0.0009089733357541263, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024638459086418152, "sampling/sampling_logp_difference/max": 1.1124024391174316, "sampling/importance_sampling_ratio/min": 0.32876816391944885, "sampling/importance_sampling_ratio/mean": 1.0097090005874634, "sampling/importance_sampling_ratio/max": 1.9459387063980103, "entropy": 0.18640629388391972, "clip_ratio/low_mean": 0.011666666949167848, "clip_ratio/low_min": 0.011666666949167848, "clip_ratio/high_mean": 0.006349399336613715, "clip_ratio/high_max": 0.006349399336613715, "clip_ratio/region_mean": 0.018016066285781562, "reward_total_mean": 0.9963726997375488, "reward_meter_mean": 0.9963726997375488, "reward_meter_std": 0.000908972229808569, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9963726997375488, "reward_total_composite_std": 0.000908972229808569} {"timestamp_utc": "2026-04-12T01:50:26Z", "mode": "train", "global_step": 2313, "epoch": 0.09290275936859863, "loss": 0.001, "grad_norm": 1.909967303276062, "learning_rate": 2.993939393939394e-06, "num_tokens": 5234365.0, "completions/mean_length": 54.875, "completions/min_length": 54.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.875, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9950602054595947, "rewards/meter/std": 0.0005074667278677225, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9950602054595947, "rewards/total_composite/std": 0.0005074667278677225, "reward": 0.9950602054595947, "reward_std": 0.0005074667278677225, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007333268411457539, "sampling/sampling_logp_difference/max": 0.7493400573730469, "sampling/importance_sampling_ratio/min": 0.47267839312553406, "sampling/importance_sampling_ratio/mean": 0.9969737529754639, "sampling/importance_sampling_ratio/max": 1.1668245792388916, "entropy": 0.02833111328072846, "clip_ratio/low_mean": 0.002314814832061529, "clip_ratio/low_min": 0.002314814832061529, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.002314814832061529, "reward_total_mean": 0.9950602054595947, "reward_meter_mean": 0.9950602054595947, "reward_meter_std": 0.0005074667278677225, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9950602054595947, "reward_total_composite_std": 0.0005074667278677225} {"timestamp_utc": "2026-04-12T01:50:31Z", "mode": "train", "global_step": 2314, "epoch": 0.09294292485038358, "loss": -0.005, "grad_norm": 1.915865182876587, "learning_rate": 2.990909090909091e-06, "num_tokens": 5237243.0, "completions/mean_length": 151.75, "completions/min_length": 149.0, "completions/max_length": 154.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 151.75, "completions/min_terminated_length": 149.0, "completions/max_terminated_length": 154.0, "rewards/meter/mean": 0.9968797564506531, "rewards/meter/std": 0.0010569265577942133, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.10101525485515594, "rewards/total_composite/mean": 0.9612998962402344, "rewards/total_composite/std": 0.10089994966983795, "reward": 0.9612998962402344, "reward_std": 0.10089994221925735, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027893707156181335, "sampling/sampling_logp_difference/max": 1.2294130325317383, "sampling/importance_sampling_ratio/min": 0.29246416687965393, "sampling/importance_sampling_ratio/mean": 1.0044018030166626, "sampling/importance_sampling_ratio/max": 1.6189171075820923, "entropy": 0.21811402402818203, "clip_ratio/low_mean": 0.0008389261784031987, "clip_ratio/low_min": 0.0008389261784031987, "clip_ratio/high_mean": 0.0221836578566581, "clip_ratio/high_max": 0.0221836578566581, "clip_ratio/region_mean": 0.0230225840350613, "reward_total_mean": 0.9612998962402344, "reward_meter_mean": 0.9968797564506531, "reward_meter_std": 0.0010569265577942133, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.10101525485515594, "reward_total_composite_mean": 0.9612998962402344, "reward_total_composite_std": 0.10089994966983795} {"timestamp_utc": "2026-04-12T01:50:36Z", "mode": "train", "global_step": 2315, "epoch": 0.09298309033216853, "loss": -0.001, "grad_norm": 2.175835132598877, "learning_rate": 2.987878787878788e-06, "num_tokens": 5239130.0, "completions/mean_length": 74.875, "completions/min_length": 73.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.875, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9990985989570618, "rewards/meter/std": 0.00011346396786393598, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990985989570618, "rewards/total_composite/std": 0.00011346396786393598, "reward": 0.9990985989570618, "reward_std": 0.00011346933752065524, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02791663445532322, "sampling/sampling_logp_difference/max": 1.1362063884735107, "sampling/importance_sampling_ratio/min": 0.32103458046913147, "sampling/importance_sampling_ratio/mean": 1.0061297416687012, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.21749908849596977, "clip_ratio/low_mean": 0.005022522644139826, "clip_ratio/low_min": 0.005022522644139826, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.005022522644139826, "reward_total_mean": 0.9990985989570618, "reward_meter_mean": 0.9990985989570618, "reward_meter_std": 0.00011346396786393598, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990985989570618, "reward_total_composite_std": 0.00011346396786393598} {"timestamp_utc": "2026-04-12T01:50:41Z", "mode": "train", "global_step": 2316, "epoch": 0.09302325581395349, "loss": 0.0011, "grad_norm": 4.3104939460754395, "learning_rate": 2.984848484848485e-06, "num_tokens": 5241562.0, "completions/mean_length": 128.0, "completions/min_length": 127.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.0, "completions/min_terminated_length": 127.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9043974876403809, "rewards/meter/std": 0.2645852863788605, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7751978635787964, "rewards/total_composite/std": 0.22678740322589874, "reward": 0.7751978635787964, "reward_std": 0.22678740322589874, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01084654126316309, "sampling/sampling_logp_difference/max": 0.9293702840805054, "sampling/importance_sampling_ratio/min": 0.3948022723197937, "sampling/importance_sampling_ratio/mean": 1.003368854522705, "sampling/importance_sampling_ratio/max": 1.5058414936065674, "entropy": 0.0738038350827992, "clip_ratio/low_mean": 0.0049212598241865635, "clip_ratio/low_min": 0.0049212598241865635, "clip_ratio/high_mean": 0.0029296875, "clip_ratio/high_max": 0.0029296875, "clip_ratio/region_mean": 0.007850947324186563, "reward_total_mean": 0.7751978635787964, "reward_meter_mean": 0.9043974876403809, "reward_meter_std": 0.2645852863788605, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7751978635787964, "reward_total_composite_std": 0.22678740322589874} {"timestamp_utc": "2026-04-12T01:50:50Z", "mode": "train", "global_step": 2317, "epoch": 0.09306342129573844, "loss": -0.0076, "grad_norm": 1.892166256904602, "learning_rate": 2.981818181818182e-06, "num_tokens": 5246963.0, "completions/mean_length": 424.125, "completions/min_length": 401.0, "completions/max_length": 450.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 424.125, "completions/min_terminated_length": 401.0, "completions/max_terminated_length": 450.0, "rewards/meter/mean": 0.992563009262085, "rewards/meter/std": 0.0028387887869030237, "rewards/count_adherence/mean": 0.6447368264198303, "rewards/count_adherence/std": 0.024363704025745392, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9372192025184631, "rewards/repeat_penalty/std": 0.03141416236758232, "rewards/total_composite/mean": 0.5997676849365234, "rewards/total_composite/std": 0.030326971784234047, "reward": 0.5997676849365234, "reward_std": 0.030326958745718002, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0303457360714674, "sampling/sampling_logp_difference/max": 1.870690107345581, "sampling/importance_sampling_ratio/min": 0.1540173441171646, "sampling/importance_sampling_ratio/mean": 1.0031834840774536, "sampling/importance_sampling_ratio/max": 1.956535816192627, "entropy": 0.2362643126398325, "clip_ratio/low_mean": 0.012263738899491727, "clip_ratio/low_min": 0.012263738899491727, "clip_ratio/high_mean": 0.009331883396953344, "clip_ratio/high_max": 0.009331883396953344, "clip_ratio/region_mean": 0.02159562229644507, "reward_total_mean": 0.5997676849365234, "reward_meter_mean": 0.992563009262085, "reward_meter_std": 0.0028387887869030237, "reward_count_adherence_mean": 0.6447368264198303, "reward_count_adherence_std": 0.024363704025745392, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9372192025184631, "reward_repeat_penalty_std": 0.03141416236758232, "reward_total_composite_mean": 0.5997676849365234, "reward_total_composite_std": 0.030326971784234047} {"timestamp_utc": "2026-04-12T01:50:55Z", "mode": "train", "global_step": 2318, "epoch": 0.0931035867775234, "loss": -0.0195, "grad_norm": 1.0140128135681152, "learning_rate": 2.9787878787878787e-06, "num_tokens": 5248971.0, "completions/mean_length": 66.0, "completions/min_length": 63.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9957113265991211, "rewards/meter/std": 0.006800828501582146, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957113265991211, "rewards/total_composite/std": 0.006800828501582146, "reward": 0.9957113265991211, "reward_std": 0.0068008191883563995, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006797808688133955, "sampling/sampling_logp_difference/max": 1.166569709777832, "sampling/importance_sampling_ratio/min": 0.3114334046840668, "sampling/importance_sampling_ratio/mean": 0.9992537498474121, "sampling/importance_sampling_ratio/max": 1.1265363693237305, "entropy": 0.037344205658882856, "clip_ratio/low_mean": 0.003968254197388887, "clip_ratio/low_min": 0.003968254197388887, "clip_ratio/high_mean": 0.0037313431967049837, "clip_ratio/high_max": 0.0037313431967049837, "clip_ratio/region_mean": 0.007699597394093871, "reward_total_mean": 0.9957113265991211, "reward_meter_mean": 0.9957113265991211, "reward_meter_std": 0.006800828501582146, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9957113265991211, "reward_total_composite_std": 0.006800828501582146} {"timestamp_utc": "2026-04-12T01:50:59Z", "mode": "train", "global_step": 2319, "epoch": 0.09314375225930835, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.9757575757575756e-06, "num_tokens": 5250619.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00042203269549645483, "sampling/sampling_logp_difference/max": 0.014573629945516586, "sampling/importance_sampling_ratio/min": 0.9997897744178772, "sampling/importance_sampling_ratio/mean": 1.0004198551177979, "sampling/importance_sampling_ratio/max": 1.0146803855895996, "entropy": 0.0041495013574603945, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:51:05Z", "mode": "train", "global_step": 2320, "epoch": 0.0931839177410933, "loss": 0.0058, "grad_norm": 1.6090915203094482, "learning_rate": 2.9727272727272733e-06, "num_tokens": 5253749.0, "completions/mean_length": 202.25, "completions/min_length": 200.0, "completions/max_length": 204.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 202.25, "completions/min_terminated_length": 200.0, "completions/max_terminated_length": 204.0, "rewards/meter/mean": 0.9957519769668579, "rewards/meter/std": 0.004873792175203562, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9204546213150024, "rewards/repeat_penalty/std": 0.03214120864868164, "rewards/total_composite/mean": 0.7856208682060242, "rewards/total_composite/std": 0.02808825671672821, "reward": 0.7856208682060242, "reward_std": 0.028088262304663658, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02219260297715664, "sampling/sampling_logp_difference/max": 3.1726443767547607, "sampling/importance_sampling_ratio/min": 0.04189267382025719, "sampling/importance_sampling_ratio/mean": 0.9995094537734985, "sampling/importance_sampling_ratio/max": 1.7268435955047607, "entropy": 0.12941910233348608, "clip_ratio/low_mean": 0.01485577883431688, "clip_ratio/low_min": 0.01485577883431688, "clip_ratio/high_mean": 0.0024999999441206455, "clip_ratio/high_max": 0.0024999999441206455, "clip_ratio/region_mean": 0.017355778778437525, "reward_total_mean": 0.7856208682060242, "reward_meter_mean": 0.9957519769668579, "reward_meter_std": 0.004873792175203562, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9204546213150024, "reward_repeat_penalty_std": 0.03214120864868164, "reward_total_composite_mean": 0.7856208682060242, "reward_total_composite_std": 0.02808825671672821} {"timestamp_utc": "2026-04-12T01:51:10Z", "mode": "train", "global_step": 2321, "epoch": 0.09322408322287826, "loss": -0.0002, "grad_norm": 0.006876571103930473, "learning_rate": 2.96969696969697e-06, "num_tokens": 5255469.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.7876332998275757, "rewards/meter/std": 1.5573279597447254e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7876332998275757, "rewards/total_composite/std": 1.5573279597447254e-05, "reward": 0.7876332998275757, "reward_std": 1.558229632792063e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0014006540877744555, "sampling/sampling_logp_difference/max": 0.3940424919128418, "sampling/importance_sampling_ratio/min": 0.6743254065513611, "sampling/importance_sampling_ratio/mean": 0.9997325539588928, "sampling/importance_sampling_ratio/max": 1.0165002346038818, "entropy": 0.004189703788142651, "clip_ratio/low_mean": 0.002314814832061529, "clip_ratio/low_min": 0.002314814832061529, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.002314814832061529, "reward_total_mean": 0.7876332998275757, "reward_meter_mean": 0.7876332998275757, "reward_meter_std": 1.5573279597447254e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7876332998275757, "reward_total_composite_std": 1.5573279597447254e-05} {"timestamp_utc": "2026-04-12T01:51:14Z", "mode": "train", "global_step": 2322, "epoch": 0.09326424870466321, "loss": 0.0375, "grad_norm": 7.9188666343688965, "learning_rate": 2.9666666666666673e-06, "num_tokens": 5257324.0, "completions/mean_length": 67.875, "completions/min_length": 61.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.875, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.7128692269325256, "rewards/meter/std": 0.2854415774345398, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.14880475401878357, "rewards/total_composite/mean": 0.5824298858642578, "rewards/total_composite/std": 0.28861746191978455, "reward": 0.5824298858642578, "reward_std": 0.28861746191978455, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07370011508464813, "sampling/sampling_logp_difference/max": 1.0788640975952148, "sampling/importance_sampling_ratio/min": 0.3399814963340759, "sampling/importance_sampling_ratio/mean": 1.0028222799301147, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3680881727486849, "clip_ratio/low_mean": 0.032118565402925014, "clip_ratio/low_min": 0.032118565402925014, "clip_ratio/high_mean": 0.04014339251443744, "clip_ratio/high_max": 0.04014339251443744, "clip_ratio/region_mean": 0.07226195791736245, "reward_total_mean": 0.5824298858642578, "reward_meter_mean": 0.7128692269325256, "reward_meter_std": 0.2854415774345398, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.14880475401878357, "reward_total_composite_mean": 0.5824298858642578, "reward_total_composite_std": 0.28861746191978455} {"timestamp_utc": "2026-04-12T01:51:19Z", "mode": "train", "global_step": 2323, "epoch": 0.09330441418644816, "loss": 0.0003, "grad_norm": 0.16411831974983215, "learning_rate": 2.963636363636364e-06, "num_tokens": 5259053.0, "completions/mean_length": 64.125, "completions/min_length": 64.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.125, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9993196725845337, "rewards/meter/std": 1.4266728612710722e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993196725845337, "rewards/total_composite/std": 1.4266728612710722e-05, "reward": 0.9993196725845337, "reward_std": 1.4269724488258362e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0027244146913290024, "sampling/sampling_logp_difference/max": 0.8163547515869141, "sampling/importance_sampling_ratio/min": 0.4420400857925415, "sampling/importance_sampling_ratio/mean": 0.9990805387496948, "sampling/importance_sampling_ratio/max": 1.0394550561904907, "entropy": 0.009595987561624497, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.001923076924867928, "clip_ratio/high_max": 0.001923076924867928, "clip_ratio/region_mean": 0.001923076924867928, "reward_total_mean": 0.9993196725845337, "reward_meter_mean": 0.9993196725845337, "reward_meter_std": 1.4266728612710722e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993196725845337, "reward_total_composite_std": 1.4266728612710722e-05} {"timestamp_utc": "2026-04-12T01:51:24Z", "mode": "train", "global_step": 2324, "epoch": 0.09334457966823312, "loss": 0.0001, "grad_norm": 3.402935028076172, "learning_rate": 2.960606060606061e-06, "num_tokens": 5261376.0, "completions/mean_length": 115.375, "completions/min_length": 112.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.375, "completions/min_terminated_length": 112.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.9961607456207275, "rewards/meter/std": 0.003092888044193387, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9961607456207275, "rewards/total_composite/std": 0.003092888044193387, "reward": 0.9961607456207275, "reward_std": 0.0030928829219192266, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027808500453829765, "sampling/sampling_logp_difference/max": 1.3722867965698242, "sampling/importance_sampling_ratio/min": 0.2535265386104584, "sampling/importance_sampling_ratio/mean": 1.0031054019927979, "sampling/importance_sampling_ratio/max": 1.6382875442504883, "entropy": 0.2400868460536003, "clip_ratio/low_mean": 0.004310344811528921, "clip_ratio/low_min": 0.004310344811528921, "clip_ratio/high_mean": 0.021577789448201656, "clip_ratio/high_max": 0.021577789448201656, "clip_ratio/region_mean": 0.025888134259730577, "reward_total_mean": 0.9961607456207275, "reward_meter_mean": 0.9961607456207275, "reward_meter_std": 0.003092888044193387, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9961607456207275, "reward_total_composite_std": 0.003092888044193387} {"timestamp_utc": "2026-04-12T01:51:29Z", "mode": "train", "global_step": 2325, "epoch": 0.09338474515001807, "loss": 0.0074, "grad_norm": 4.780300140380859, "learning_rate": 2.957575757575758e-06, "num_tokens": 5263331.0, "completions/mean_length": 67.375, "completions/min_length": 65.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.375, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9977126121520996, "rewards/meter/std": 0.0028336539398878813, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977126121520996, "rewards/total_composite/std": 0.0028336539398878813, "reward": 0.9977126121520996, "reward_std": 0.0028336546383798122, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.049458228051662445, "sampling/sampling_logp_difference/max": 1.1433992385864258, "sampling/importance_sampling_ratio/min": 0.31873372197151184, "sampling/importance_sampling_ratio/mean": 1.0002453327178955, "sampling/importance_sampling_ratio/max": 1.6862565279006958, "entropy": 0.3584112077951431, "clip_ratio/low_mean": 0.009615384973585606, "clip_ratio/low_min": 0.009615384973585606, "clip_ratio/high_mean": 0.02963537711184472, "clip_ratio/high_max": 0.02963537711184472, "clip_ratio/region_mean": 0.039250762085430324, "reward_total_mean": 0.9977126121520996, "reward_meter_mean": 0.9977126121520996, "reward_meter_std": 0.0028336539398878813, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9977126121520996, "reward_total_composite_std": 0.0028336539398878813} {"timestamp_utc": "2026-04-12T01:51:34Z", "mode": "train", "global_step": 2326, "epoch": 0.09342491063180303, "loss": -0.0182, "grad_norm": 3.8735227584838867, "learning_rate": 2.954545454545455e-06, "num_tokens": 5265267.0, "completions/mean_length": 78.0, "completions/min_length": 75.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.0, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9950919151306152, "rewards/meter/std": 0.007245616987347603, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9950919151306152, "rewards/total_composite/std": 0.007245616987347603, "reward": 0.9950919151306152, "reward_std": 0.007245623506605625, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01848999224603176, "sampling/sampling_logp_difference/max": 0.8240165710449219, "sampling/importance_sampling_ratio/min": 0.4386661946773529, "sampling/importance_sampling_ratio/mean": 1.0039349794387817, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.16736672818660736, "clip_ratio/low_mean": 0.0033333334140479565, "clip_ratio/low_min": 0.0033333334140479565, "clip_ratio/high_mean": 0.011138874455355108, "clip_ratio/high_max": 0.011138874455355108, "clip_ratio/region_mean": 0.014472207869403064, "reward_total_mean": 0.9950919151306152, "reward_meter_mean": 0.9950919151306152, "reward_meter_std": 0.007245616987347603, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9950919151306152, "reward_total_composite_std": 0.007245616987347603} {"timestamp_utc": "2026-04-12T01:51:39Z", "mode": "train", "global_step": 2327, "epoch": 0.09346507611358798, "loss": -0.0017, "grad_norm": 7.240959167480469, "learning_rate": 2.951515151515152e-06, "num_tokens": 5266851.0, "completions/mean_length": 39.0, "completions/min_length": 37.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.0, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.9987367391586304, "rewards/meter/std": 0.0013684039004147053, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9987367391586304, "rewards/total_composite/std": 0.0013684039004147053, "reward": 0.9987367391586304, "reward_std": 0.0013684096047654748, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.032071422785520554, "sampling/sampling_logp_difference/max": 0.8518095016479492, "sampling/importance_sampling_ratio/min": 0.42664220929145813, "sampling/importance_sampling_ratio/mean": 1.003873348236084, "sampling/importance_sampling_ratio/max": 1.3673170804977417, "entropy": 0.19926494918763638, "clip_ratio/low_mean": 0.00937500037252903, "clip_ratio/low_min": 0.00937500037252903, "clip_ratio/high_mean": 0.009627727791666985, "clip_ratio/high_max": 0.009627727791666985, "clip_ratio/region_mean": 0.019002728164196014, "reward_total_mean": 0.9987367391586304, "reward_meter_mean": 0.9987367391586304, "reward_meter_std": 0.0013684039004147053, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9987367391586304, "reward_total_composite_std": 0.0013684039004147053} {"timestamp_utc": "2026-04-12T01:51:44Z", "mode": "train", "global_step": 2328, "epoch": 0.09350524159537293, "loss": 0.0007, "grad_norm": 0.9024956822395325, "learning_rate": 2.9484848484848488e-06, "num_tokens": 5269464.0, "completions/mean_length": 159.625, "completions/min_length": 158.0, "completions/max_length": 160.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 159.625, "completions/min_terminated_length": 158.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.9978625774383545, "rewards/meter/std": 3.399623528821394e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.819444477558136, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.8176930546760559, "rewards/total_composite/std": 0.057384345680475235, "reward": 0.8176930546760559, "reward_std": 0.05738433822989464, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010679478757083416, "sampling/sampling_logp_difference/max": 1.0905389785766602, "sampling/importance_sampling_ratio/min": 0.3360353410243988, "sampling/importance_sampling_ratio/mean": 0.9999021291732788, "sampling/importance_sampling_ratio/max": 1.274688720703125, "entropy": 0.060957128182053566, "clip_ratio/low_mean": 0.0007812500116415322, "clip_ratio/low_min": 0.0007812500116415322, "clip_ratio/high_mean": 0.0031250000465661287, "clip_ratio/high_max": 0.0031250000465661287, "clip_ratio/region_mean": 0.003906250058207661, "reward_total_mean": 0.8176930546760559, "reward_meter_mean": 0.9978625774383545, "reward_meter_std": 3.399623528821394e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.819444477558136, "reward_repeat_penalty_std": 0.05750546231865883, "reward_total_composite_mean": 0.8176930546760559, "reward_total_composite_std": 0.057384345680475235} {"timestamp_utc": "2026-04-12T01:51:49Z", "mode": "train", "global_step": 2329, "epoch": 0.09354540707715789, "loss": 0.0043, "grad_norm": 3.704277992248535, "learning_rate": 2.9454545454545456e-06, "num_tokens": 5271316.0, "completions/mean_length": 66.5, "completions/min_length": 65.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.5, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9816258549690247, "rewards/meter/std": 0.004315865226089954, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9816258549690247, "rewards/total_composite/std": 0.004315865226089954, "reward": 0.9816258549690247, "reward_std": 0.004315854981541634, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022112296894192696, "sampling/sampling_logp_difference/max": 1.1111278533935547, "sampling/importance_sampling_ratio/min": 0.3291874825954437, "sampling/importance_sampling_ratio/mean": 1.0040234327316284, "sampling/importance_sampling_ratio/max": 1.3998769521713257, "entropy": 0.1675149966031313, "clip_ratio/low_mean": 0.013287583249621093, "clip_ratio/low_min": 0.013287583249621093, "clip_ratio/high_mean": 0.0037313431967049837, "clip_ratio/high_max": 0.0037313431967049837, "clip_ratio/region_mean": 0.017018926446326077, "reward_total_mean": 0.9816258549690247, "reward_meter_mean": 0.9816258549690247, "reward_meter_std": 0.004315865226089954, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9816258549690247, "reward_total_composite_std": 0.004315865226089954} {"timestamp_utc": "2026-04-12T01:51:53Z", "mode": "train", "global_step": 2330, "epoch": 0.09358557255894284, "loss": 0.001, "grad_norm": 4.507577419281006, "learning_rate": 2.942424242424243e-06, "num_tokens": 5272995.0, "completions/mean_length": 67.875, "completions/min_length": 67.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.875, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9815051555633545, "rewards/meter/std": 0.004705195315182209, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9815051555633545, "rewards/total_composite/std": 0.004705195315182209, "reward": 0.9815051555633545, "reward_std": 0.004705206956714392, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015049697831273079, "sampling/sampling_logp_difference/max": 0.6336269378662109, "sampling/importance_sampling_ratio/min": 0.5306636691093445, "sampling/importance_sampling_ratio/mean": 1.0061925649642944, "sampling/importance_sampling_ratio/max": 1.3674873113632202, "entropy": 0.0950354440137744, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/high_mean": 0.0036764706019312143, "clip_ratio/high_max": 0.0036764706019312143, "clip_ratio/region_mean": 0.005542142200283706, "reward_total_mean": 0.9815051555633545, "reward_meter_mean": 0.9815051555633545, "reward_meter_std": 0.004705195315182209, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9815051555633545, "reward_total_composite_std": 0.004705195315182209} {"timestamp_utc": "2026-04-12T01:51:58Z", "mode": "train", "global_step": 2331, "epoch": 0.0936257380407278, "loss": 0.0002, "grad_norm": 3.0374279022216797, "learning_rate": 2.9393939393939397e-06, "num_tokens": 5274705.0, "completions/mean_length": 65.75, "completions/min_length": 64.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.75, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9988436698913574, "rewards/meter/std": 0.00024908926570788026, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9988436698913574, "rewards/total_composite/std": 0.00024908926570788026, "reward": 0.9988436698913574, "reward_std": 0.000249094155151397, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04332128167152405, "sampling/sampling_logp_difference/max": 1.0077180862426758, "sampling/importance_sampling_ratio/min": 0.3650510609149933, "sampling/importance_sampling_ratio/mean": 1.0106713771820068, "sampling/importance_sampling_ratio/max": 1.7954013347625732, "entropy": 0.3256809823215008, "clip_ratio/low_mean": 0.020951705053448677, "clip_ratio/low_min": 0.020951705053448677, "clip_ratio/high_mean": 0.018828062573447824, "clip_ratio/high_max": 0.018828062573447824, "clip_ratio/region_mean": 0.0397797676268965, "reward_total_mean": 0.9988436698913574, "reward_meter_mean": 0.9988436698913574, "reward_meter_std": 0.00024908926570788026, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9988436698913574, "reward_total_composite_std": 0.00024908926570788026} {"timestamp_utc": "2026-04-12T01:52:02Z", "mode": "train", "global_step": 2332, "epoch": 0.09366590352251275, "loss": -0.0003, "grad_norm": 0.5694947242736816, "learning_rate": 2.9363636363636365e-06, "num_tokens": 5276497.0, "completions/mean_length": 67.0, "completions/min_length": 67.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.998113214969635, "rewards/meter/std": 3.099795139860362e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998113214969635, "rewards/total_composite/std": 3.099795139860362e-05, "reward": 0.998113214969635, "reward_std": 3.100236426689662e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007214725017547607, "sampling/sampling_logp_difference/max": 0.7471715211868286, "sampling/importance_sampling_ratio/min": 0.4737045168876648, "sampling/importance_sampling_ratio/mean": 1.002618670463562, "sampling/importance_sampling_ratio/max": 1.6667267084121704, "entropy": 0.05256783217191696, "clip_ratio/low_mean": 0.005597014795057476, "clip_ratio/low_min": 0.005597014795057476, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.005597014795057476, "reward_total_mean": 0.998113214969635, "reward_meter_mean": 0.998113214969635, "reward_meter_std": 3.099795139860362e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.998113214969635, "reward_total_composite_std": 3.099795139860362e-05} {"timestamp_utc": "2026-04-12T01:52:07Z", "mode": "train", "global_step": 2333, "epoch": 0.0937060690042977, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.9333333333333338e-06, "num_tokens": 5278425.0, "completions/mean_length": 80.0, "completions/min_length": 80.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.0, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "reward": 0.7190229296684265, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0004051509313285351, "sampling/sampling_logp_difference/max": 0.01761876977980137, "sampling/importance_sampling_ratio/min": 0.9928526878356934, "sampling/importance_sampling_ratio/mean": 1.0003622770309448, "sampling/importance_sampling_ratio/max": 1.0177748203277588, "entropy": 0.0035651897196657956, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7190229296684265, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:52:13Z", "mode": "train", "global_step": 2334, "epoch": 0.09374623448608266, "loss": -0.0067, "grad_norm": 8.575316429138184, "learning_rate": 2.9303030303030306e-06, "num_tokens": 5280271.0, "completions/mean_length": 66.75, "completions/min_length": 64.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.75, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9969208836555481, "rewards/meter/std": 0.0032654430251568556, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9969208836555481, "rewards/total_composite/std": 0.0032654430251568556, "reward": 0.9969208836555481, "reward_std": 0.003265430685132742, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012270020321011543, "sampling/sampling_logp_difference/max": 1.2344985008239746, "sampling/importance_sampling_ratio/min": 0.29098066687583923, "sampling/importance_sampling_ratio/mean": 0.9968401193618774, "sampling/importance_sampling_ratio/max": 1.4511247873306274, "entropy": 0.0710354926995933, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.011166593292728066, "clip_ratio/high_max": 0.011166593292728066, "clip_ratio/region_mean": 0.011166593292728066, "reward_total_mean": 0.9969208836555481, "reward_meter_mean": 0.9969208836555481, "reward_meter_std": 0.0032654430251568556, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9969208836555481, "reward_total_composite_std": 0.0032654430251568556} {"timestamp_utc": "2026-04-12T01:52:18Z", "mode": "train", "global_step": 2335, "epoch": 0.09378639996786761, "loss": 0.0011, "grad_norm": 2.4751315116882324, "learning_rate": 2.9272727272727274e-06, "num_tokens": 5281911.0, "completions/mean_length": 55.0, "completions/min_length": 55.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.0, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9950971603393555, "rewards/meter/std": 0.0004900125786662102, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9950971603393555, "rewards/total_composite/std": 0.0004900125786662102, "reward": 0.9950971603393555, "reward_std": 0.0004900065250694752, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005754906218498945, "sampling/sampling_logp_difference/max": 0.3944000005722046, "sampling/importance_sampling_ratio/min": 0.6740844249725342, "sampling/importance_sampling_ratio/mean": 0.9981791973114014, "sampling/importance_sampling_ratio/max": 1.2244776487350464, "entropy": 0.022867268649861217, "clip_ratio/low_mean": 0.0022727272007614374, "clip_ratio/low_min": 0.0022727272007614374, "clip_ratio/high_mean": 0.0022727272007614374, "clip_ratio/high_max": 0.0022727272007614374, "clip_ratio/region_mean": 0.004545454401522875, "reward_total_mean": 0.9950971603393555, "reward_meter_mean": 0.9950971603393555, "reward_meter_std": 0.0004900125786662102, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9950971603393555, "reward_total_composite_std": 0.0004900125786662102} {"timestamp_utc": "2026-04-12T01:52:22Z", "mode": "train", "global_step": 2336, "epoch": 0.09382656544965257, "loss": 0.0022, "grad_norm": 2.559536933898926, "learning_rate": 2.9242424242424243e-06, "num_tokens": 5283616.0, "completions/mean_length": 67.125, "completions/min_length": 67.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.125, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9981128573417664, "rewards/meter/std": 5.023560152039863e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981128573417664, "rewards/total_composite/std": 5.023560152039863e-05, "reward": 0.9981128573417664, "reward_std": 5.024260462960228e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004402663093060255, "sampling/sampling_logp_difference/max": 0.3112151622772217, "sampling/importance_sampling_ratio/min": 0.8633265495300293, "sampling/importance_sampling_ratio/mean": 1.00238835811615, "sampling/importance_sampling_ratio/max": 1.3650829792022705, "entropy": 0.04221090069040656, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9981128573417664, "reward_meter_mean": 0.9981128573417664, "reward_meter_std": 5.023560152039863e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981128573417664, "reward_total_composite_std": 5.023560152039863e-05} {"timestamp_utc": "2026-04-12T01:52:28Z", "mode": "train", "global_step": 2337, "epoch": 0.09386673093143752, "loss": 0.0008, "grad_norm": 1.5094997882843018, "learning_rate": 2.9212121212121215e-06, "num_tokens": 5287016.0, "completions/mean_length": 207.0, "completions/min_length": 206.0, "completions/max_length": 208.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 207.0, "completions/min_terminated_length": 206.0, "completions/max_terminated_length": 208.0, "rewards/meter/mean": 0.9978703260421753, "rewards/meter/std": 3.329759420012124e-05, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8020833730697632, "rewards/repeat_penalty/std": 0.10853918641805649, "rewards/total_composite/mean": 0.686035692691803, "rewards/total_composite/std": 0.09283558279275894, "reward": 0.686035692691803, "reward_std": 0.09283558279275894, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013423108495771885, "sampling/sampling_logp_difference/max": 1.0543880462646484, "sampling/importance_sampling_ratio/min": 0.34840553998947144, "sampling/importance_sampling_ratio/mean": 1.0003283023834229, "sampling/importance_sampling_ratio/max": 1.4435890913009644, "entropy": 0.09781584423035383, "clip_ratio/low_mean": 0.007231889001559466, "clip_ratio/low_min": 0.007231889001559466, "clip_ratio/high_mean": 0.0036231884150765836, "clip_ratio/high_max": 0.0036231884150765836, "clip_ratio/region_mean": 0.01085507741663605, "reward_total_mean": 0.686035692691803, "reward_meter_mean": 0.9978703260421753, "reward_meter_std": 3.329759420012124e-05, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8020833730697632, "reward_repeat_penalty_std": 0.10853918641805649, "reward_total_composite_mean": 0.686035692691803, "reward_total_composite_std": 0.09283558279275894} {"timestamp_utc": "2026-04-12T01:52:33Z", "mode": "train", "global_step": 2338, "epoch": 0.09390689641322247, "loss": -0.0001, "grad_norm": 0.4725949764251709, "learning_rate": 2.9181818181818183e-06, "num_tokens": 5288928.0, "completions/mean_length": 67.0, "completions/min_length": 67.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981162548065186, "rewards/meter/std": 2.446455619065091e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981162548065186, "rewards/total_composite/std": 2.446455619065091e-05, "reward": 0.9981162548065186, "reward_std": 2.446935832267627e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006431423127651215, "sampling/sampling_logp_difference/max": 0.8494582176208496, "sampling/importance_sampling_ratio/min": 0.42764657735824585, "sampling/importance_sampling_ratio/mean": 0.9997783899307251, "sampling/importance_sampling_ratio/max": 1.1632411479949951, "entropy": 0.04713014606386423, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9981162548065186, "reward_meter_mean": 0.9981162548065186, "reward_meter_std": 2.446455619065091e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981162548065186, "reward_total_composite_std": 2.446455619065091e-05} {"timestamp_utc": "2026-04-12T01:52:38Z", "mode": "train", "global_step": 2339, "epoch": 0.09394706189500743, "loss": 0.0017, "grad_norm": 0.770327091217041, "learning_rate": 2.915151515151515e-06, "num_tokens": 5291330.0, "completions/mean_length": 129.25, "completions/min_length": 129.0, "completions/max_length": 130.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 129.25, "completions/min_terminated_length": 129.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.9979046583175659, "rewards/meter/std": 4.928474663756788e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8553467988967896, "rewards/total_composite/std": 4.225334123475477e-05, "reward": 0.8553467988967896, "reward_std": 4.226414966979064e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010605924762785435, "sampling/sampling_logp_difference/max": 1.4333992004394531, "sampling/importance_sampling_ratio/min": 0.23849685490131378, "sampling/importance_sampling_ratio/mean": 0.9997851848602295, "sampling/importance_sampling_ratio/max": 1.9889717102050781, "entropy": 0.040207551792263985, "clip_ratio/low_mean": 0.001923076924867928, "clip_ratio/low_min": 0.001923076924867928, "clip_ratio/high_mean": 0.00484496122226119, "clip_ratio/high_max": 0.00484496122226119, "clip_ratio/region_mean": 0.0067680381471291184, "reward_total_mean": 0.8553467988967896, "reward_meter_mean": 0.9979046583175659, "reward_meter_std": 4.928474663756788e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8553467988967896, "reward_total_composite_std": 4.225334123475477e-05} {"timestamp_utc": "2026-04-12T01:52:46Z", "mode": "train", "global_step": 2340, "epoch": 0.09398722737679238, "loss": -0.0245, "grad_norm": 1.771281361579895, "learning_rate": 2.9121212121212124e-06, "num_tokens": 5295161.0, "completions/mean_length": 262.875, "completions/min_length": 252.0, "completions/max_length": 289.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 262.875, "completions/min_terminated_length": 252.0, "completions/max_terminated_length": 289.0, "rewards/meter/mean": 0.9991946220397949, "rewards/meter/std": 0.00014320346235763282, "rewards/count_adherence/mean": 0.8055555820465088, "rewards/count_adherence/std": 0.05143444985151291, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8057692050933838, "rewards/repeat_penalty/std": 0.08968339115381241, "rewards/total_composite/mean": 0.6484043002128601, "rewards/total_composite/std": 0.07946042716503143, "reward": 0.6484043002128601, "reward_std": 0.07946040481328964, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017933662980794907, "sampling/sampling_logp_difference/max": 1.0136165618896484, "sampling/importance_sampling_ratio/min": 0.3629041314125061, "sampling/importance_sampling_ratio/mean": 1.00568687915802, "sampling/importance_sampling_ratio/max": 1.6187031269073486, "entropy": 0.18291304260492325, "clip_ratio/low_mean": 0.004446686594747007, "clip_ratio/low_min": 0.004446686594747007, "clip_ratio/high_mean": 0.011448833451140672, "clip_ratio/high_max": 0.011448833451140672, "clip_ratio/region_mean": 0.01589552004588768, "reward_total_mean": 0.6484043002128601, "reward_meter_mean": 0.9991946220397949, "reward_meter_std": 0.00014320346235763282, "reward_count_adherence_mean": 0.8055555820465088, "reward_count_adherence_std": 0.05143444985151291, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8057692050933838, "reward_repeat_penalty_std": 0.08968339115381241, "reward_total_composite_mean": 0.6484043002128601, "reward_total_composite_std": 0.07946042716503143} {"timestamp_utc": "2026-04-12T01:52:53Z", "mode": "train", "global_step": 2341, "epoch": 0.09402739285857734, "loss": -0.0113, "grad_norm": 1.6412862539291382, "learning_rate": 2.9090909090909093e-06, "num_tokens": 5299382.0, "completions/mean_length": 265.625, "completions/min_length": 250.0, "completions/max_length": 290.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 265.625, "completions/min_terminated_length": 250.0, "completions/max_terminated_length": 290.0, "rewards/meter/mean": 0.9990073442459106, "rewards/meter/std": 0.0001355017302557826, "rewards/count_adherence/mean": 0.737500011920929, "rewards/count_adherence/std": 0.05175492912530899, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8301282525062561, "rewards/repeat_penalty/std": 0.06518138945102692, "rewards/total_composite/mean": 0.6096493005752563, "rewards/total_composite/std": 0.03883464261889458, "reward": 0.6096493005752563, "reward_std": 0.03883464261889458, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022991513833403587, "sampling/sampling_logp_difference/max": 2.072540044784546, "sampling/importance_sampling_ratio/min": 0.30119454860687256, "sampling/importance_sampling_ratio/mean": 1.0052927732467651, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.20100159756839275, "clip_ratio/low_mean": 0.011894350522197783, "clip_ratio/low_min": 0.011894350522197783, "clip_ratio/high_mean": 0.00551724131219089, "clip_ratio/high_max": 0.00551724131219089, "clip_ratio/region_mean": 0.017411591834388673, "reward_total_mean": 0.6096493005752563, "reward_meter_mean": 0.9990073442459106, "reward_meter_std": 0.0001355017302557826, "reward_count_adherence_mean": 0.737500011920929, "reward_count_adherence_std": 0.05175492912530899, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8301282525062561, "reward_repeat_penalty_std": 0.06518138945102692, "reward_total_composite_mean": 0.6096493005752563, "reward_total_composite_std": 0.03883464261889458} {"timestamp_utc": "2026-04-12T01:52:57Z", "mode": "train", "global_step": 2342, "epoch": 0.09406755834036229, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.906060606060606e-06, "num_tokens": 5301206.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00041229455382563174, "sampling/sampling_logp_difference/max": 0.0067067258059978485, "sampling/importance_sampling_ratio/min": 0.9993113875389099, "sampling/importance_sampling_ratio/mean": 1.000400185585022, "sampling/importance_sampling_ratio/max": 1.0067293643951416, "entropy": 0.003808505309280008, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:53:02Z", "mode": "train", "global_step": 2343, "epoch": 0.09410772382214724, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.903030303030303e-06, "num_tokens": 5303262.0, "completions/mean_length": 80.0, "completions/min_length": 80.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.0, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "reward": 0.7190229296684265, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00038630847120657563, "sampling/sampling_logp_difference/max": 0.01495426706969738, "sampling/importance_sampling_ratio/min": 0.9851569533348083, "sampling/importance_sampling_ratio/mean": 1.0003089904785156, "sampling/importance_sampling_ratio/max": 1.0098844766616821, "entropy": 0.0037699798413086683, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7190229296684265, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:53:07Z", "mode": "train", "global_step": 2344, "epoch": 0.0941478893039322, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.9e-06, "num_tokens": 5305230.0, "completions/mean_length": 80.0, "completions/min_length": 80.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.0, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "reward": 0.7190229296684265, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00038092854083515704, "sampling/sampling_logp_difference/max": 0.01068283524364233, "sampling/importance_sampling_ratio/min": 0.989374041557312, "sampling/importance_sampling_ratio/mean": 1.000275731086731, "sampling/importance_sampling_ratio/max": 1.0104482173919678, "entropy": 0.0038317264115903527, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7190229296684265, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:53:11Z", "mode": "train", "global_step": 2345, "epoch": 0.09418805478571715, "loss": 0.0324, "grad_norm": 9.732110023498535, "learning_rate": 2.896969696969697e-06, "num_tokens": 5306953.0, "completions/mean_length": 64.375, "completions/min_length": 56.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.375, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.7194586992263794, "rewards/meter/std": 0.26375213265419006, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.7001276016235352, "rewards/total_composite/std": 0.2649116516113281, "reward": 0.7001276016235352, "reward_std": 0.26491162180900574, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06900904327630997, "sampling/sampling_logp_difference/max": 2.044987678527832, "sampling/importance_sampling_ratio/min": 0.12938177585601807, "sampling/importance_sampling_ratio/mean": 1.0110507011413574, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3568100146949291, "clip_ratio/low_mean": 0.023096519173122942, "clip_ratio/low_min": 0.023096519173122942, "clip_ratio/high_mean": 0.029499810189008713, "clip_ratio/high_max": 0.029499810189008713, "clip_ratio/region_mean": 0.052596329362131655, "reward_total_mean": 0.7001276016235352, "reward_meter_mean": 0.7194586992263794, "reward_meter_std": 0.26375213265419006, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.7001276016235352, "reward_total_composite_std": 0.2649116516113281} {"timestamp_utc": "2026-04-12T01:53:18Z", "mode": "train", "global_step": 2346, "epoch": 0.0942282202675021, "loss": -0.0015, "grad_norm": 3.087636947631836, "learning_rate": 2.893939393939394e-06, "num_tokens": 5310264.0, "completions/mean_length": 237.875, "completions/min_length": 228.0, "completions/max_length": 248.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 237.875, "completions/min_terminated_length": 228.0, "completions/max_terminated_length": 248.0, "rewards/meter/mean": 0.9981396794319153, "rewards/meter/std": 0.0003048716171178967, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.942307710647583, "rewards/repeat_penalty/std": 0.07962294667959213, "rewards/total_composite/mean": 0.9405449032783508, "rewards/total_composite/std": 0.07935542613267899, "reward": 0.9405449032783508, "reward_std": 0.07935541868209839, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0508662573993206, "sampling/sampling_logp_difference/max": 5.533112049102783, "sampling/importance_sampling_ratio/min": 0.003953665494918823, "sampling/importance_sampling_ratio/mean": 1.0058072805404663, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4049236960709095, "clip_ratio/low_mean": 0.014375669648870826, "clip_ratio/low_min": 0.014375669648870826, "clip_ratio/high_mean": 0.02505582571029663, "clip_ratio/high_max": 0.02505582571029663, "clip_ratio/region_mean": 0.03943149535916746, "reward_total_mean": 0.9405449032783508, "reward_meter_mean": 0.9981396794319153, "reward_meter_std": 0.0003048716171178967, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.942307710647583, "reward_repeat_penalty_std": 0.07962294667959213, "reward_total_composite_mean": 0.9405449032783508, "reward_total_composite_std": 0.07935542613267899} {"timestamp_utc": "2026-04-12T01:53:23Z", "mode": "train", "global_step": 2347, "epoch": 0.09426838574928706, "loss": -0.0006, "grad_norm": 3.2611429691314697, "learning_rate": 2.8909090909090907e-06, "num_tokens": 5312388.0, "completions/mean_length": 100.5, "completions/min_length": 97.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.5, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.9978408217430115, "rewards/meter/std": 0.0007257388206198812, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978408217430115, "rewards/total_composite/std": 0.0007257388206198812, "reward": 0.9978408217430115, "reward_std": 0.0007257444667629898, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.056275609880685806, "sampling/sampling_logp_difference/max": 2.3127644062042236, "sampling/importance_sampling_ratio/min": 0.09898723661899567, "sampling/importance_sampling_ratio/mean": 1.0071477890014648, "sampling/importance_sampling_ratio/max": 1.796935796737671, "entropy": 0.4347502291202545, "clip_ratio/low_mean": 0.017315812641754746, "clip_ratio/low_min": 0.017315812641754746, "clip_ratio/high_mean": 0.031063538044691086, "clip_ratio/high_max": 0.031063538044691086, "clip_ratio/region_mean": 0.04837935068644583, "reward_total_mean": 0.9978408217430115, "reward_meter_mean": 0.9978408217430115, "reward_meter_std": 0.0007257388206198812, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9978408217430115, "reward_total_composite_std": 0.0007257388206198812} {"timestamp_utc": "2026-04-12T01:53:29Z", "mode": "train", "global_step": 2348, "epoch": 0.09430855123107201, "loss": -0.0016, "grad_norm": 2.218722105026245, "learning_rate": 2.8878787878787884e-06, "num_tokens": 5315212.0, "completions/mean_length": 165.0, "completions/min_length": 163.0, "completions/max_length": 167.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 165.0, "completions/min_terminated_length": 163.0, "completions/max_terminated_length": 167.0, "rewards/meter/mean": 0.9987882375717163, "rewards/meter/std": 0.0007174185593612492, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.930555522441864, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.9294023513793945, "rewards/total_composite/std": 0.05695149675011635, "reward": 0.9294023513793945, "reward_std": 0.05695149302482605, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018433764576911926, "sampling/sampling_logp_difference/max": 1.2507028579711914, "sampling/importance_sampling_ratio/min": 0.28630349040031433, "sampling/importance_sampling_ratio/mean": 1.0003210306167603, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10382581874728203, "clip_ratio/low_mean": 0.00988566328305751, "clip_ratio/low_min": 0.00988566328305751, "clip_ratio/high_mean": 0.003007704159244895, "clip_ratio/high_max": 0.003007704159244895, "clip_ratio/region_mean": 0.012893367442302406, "reward_total_mean": 0.9294023513793945, "reward_meter_mean": 0.9987882375717163, "reward_meter_std": 0.0007174185593612492, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.930555522441864, "reward_repeat_penalty_std": 0.05750546231865883, "reward_total_composite_mean": 0.9294023513793945, "reward_total_composite_std": 0.05695149675011635} {"timestamp_utc": "2026-04-12T01:53:34Z", "mode": "train", "global_step": 2349, "epoch": 0.09434871671285697, "loss": -0.0002, "grad_norm": 2.1028835773468018, "learning_rate": 2.884848484848485e-06, "num_tokens": 5317189.0, "completions/mean_length": 83.125, "completions/min_length": 83.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.125, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.9964075684547424, "rewards/meter/std": 0.0002765149693004787, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9964075684547424, "rewards/total_composite/std": 0.0002765149693004787, "reward": 0.9964075684547424, "reward_std": 0.0002765233220998198, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014425117522478104, "sampling/sampling_logp_difference/max": 0.9964022636413574, "sampling/importance_sampling_ratio/min": 0.3692053556442261, "sampling/importance_sampling_ratio/mean": 1.00690495967865, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.04944844264537096, "clip_ratio/low_mean": 0.013554216362535954, "clip_ratio/low_min": 0.013554216362535954, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.013554216362535954, "reward_total_mean": 0.9964075684547424, "reward_meter_mean": 0.9964075684547424, "reward_meter_std": 0.0002765149693004787, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9964075684547424, "reward_total_composite_std": 0.0002765149693004787} {"timestamp_utc": "2026-04-12T01:53:41Z", "mode": "train", "global_step": 2350, "epoch": 0.09438888219464192, "loss": -0.0123, "grad_norm": 2.5682480335235596, "learning_rate": 2.8818181818181824e-06, "num_tokens": 5320279.0, "completions/mean_length": 193.25, "completions/min_length": 185.0, "completions/max_length": 203.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 193.25, "completions/min_terminated_length": 185.0, "completions/max_terminated_length": 203.0, "rewards/meter/mean": 0.9980719089508057, "rewards/meter/std": 0.00038988571031950414, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9659091234207153, "rewards/repeat_penalty/std": 0.06763853132724762, "rewards/total_composite/mean": 0.9640417098999023, "rewards/total_composite/std": 0.06744071841239929, "reward": 0.9640417098999023, "reward_std": 0.06744074076414108, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05063340812921524, "sampling/sampling_logp_difference/max": 1.2536392211914062, "sampling/importance_sampling_ratio/min": 0.2854640483856201, "sampling/importance_sampling_ratio/mean": 1.0081247091293335, "sampling/importance_sampling_ratio/max": 1.9953590631484985, "entropy": 0.40191996842622757, "clip_ratio/low_mean": 0.004700823919847608, "clip_ratio/low_min": 0.004700823919847608, "clip_ratio/high_mean": 0.03450171835720539, "clip_ratio/high_max": 0.03450171835720539, "clip_ratio/region_mean": 0.039202542277053, "reward_total_mean": 0.9640417098999023, "reward_meter_mean": 0.9980719089508057, "reward_meter_std": 0.00038988571031950414, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9659091234207153, "reward_repeat_penalty_std": 0.06763853132724762, "reward_total_composite_mean": 0.9640417098999023, "reward_total_composite_std": 0.06744071841239929} {"timestamp_utc": "2026-04-12T01:54:49Z", "mode": "eval", "global_step": 2350, "epoch": 0.09438888219464192, "eval_loss": NaN, "eval_runtime": 67.6666, "eval_samples_per_second": 1.537, "eval_steps_per_second": 0.192, "eval_num_tokens": 5320279.0, "eval_completions/mean_length": 194.6346153846154, "eval_completions/min_length": 61.30769230769231, "eval_completions/max_length": 349.3076923076923, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 194.6346153846154, "eval_completions/min_terminated_length": 61.30769230769231, "eval_completions/max_terminated_length": 349.3076923076923, "eval_rewards/meter/mean": 0.7545687923064599, "eval_rewards/meter/std": 0.3586408014480884, "eval_rewards/count_adherence/mean": 0.9230444156206571, "eval_rewards/count_adherence/std": 0.10953088907095102, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/repeat_penalty/mean": 0.885237611257113, "eval_rewards/repeat_penalty/std": 0.1111061366704794, "eval_rewards/total_composite/mean": 0.6291824304140531, "eval_rewards/total_composite/std": 0.33865179121494293, "eval_reward": 0.6291824304140531, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.021095395159835998, "eval_sampling/sampling_logp_difference/max": 1.0696188119741588, "eval_sampling/importance_sampling_ratio/min": 0.3466451099285713, "eval_sampling/importance_sampling_ratio/mean": 1.0047486470295832, "eval_sampling/importance_sampling_ratio/max": 1.4322827320832472, "eval_entropy": 0.21742234894862542, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6291824304140531, "eval_reward_meter_mean": 0.7545687923064599, "eval_reward_meter_std": 0.3586408014480884, "eval_reward_count_adherence_mean": 0.9230444156206571, "eval_reward_count_adherence_std": 0.10953088907095102, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_repeat_penalty_mean": 0.885237611257113, "eval_reward_repeat_penalty_std": 0.1111061366704794, "eval_reward_total_composite_mean": 0.6291824304140531, "eval_reward_total_composite_std": 0.33865179121494293} {"timestamp_utc": "2026-04-12T01:54:56Z", "mode": "train", "global_step": 2351, "epoch": 0.09442904767642687, "loss": 0.0052, "grad_norm": 4.469809532165527, "learning_rate": 2.8787878787878793e-06, "num_tokens": 5322067.0, "completions/mean_length": 75.5, "completions/min_length": 73.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.5, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.998935878276825, "rewards/meter/std": 0.0006324206478893757, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998935878276825, "rewards/total_composite/std": 0.0006324206478893757, "reward": 0.998935878276825, "reward_std": 0.0006324192509055138, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030197838321328163, "sampling/sampling_logp_difference/max": 1.0953636169433594, "sampling/importance_sampling_ratio/min": 0.33441799879074097, "sampling/importance_sampling_ratio/mean": 1.0036641359329224, "sampling/importance_sampling_ratio/max": 1.4723999500274658, "entropy": 0.20778764225542545, "clip_ratio/low_mean": 0.008248508209362626, "clip_ratio/low_min": 0.008248508209362626, "clip_ratio/high_mean": 0.01327340246643871, "clip_ratio/high_max": 0.01327340246643871, "clip_ratio/region_mean": 0.021521910675801337, "reward_total_mean": 0.998935878276825, "reward_meter_mean": 0.998935878276825, "reward_meter_std": 0.0006324206478893757, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.998935878276825, "reward_total_composite_std": 0.0006324206478893757} {"timestamp_utc": "2026-04-12T01:55:00Z", "mode": "train", "global_step": 2352, "epoch": 0.09446921315821183, "loss": -0.0028, "grad_norm": 2.229356288909912, "learning_rate": 2.875757575757576e-06, "num_tokens": 5323867.0, "completions/mean_length": 55.0, "completions/min_length": 55.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.0, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9961405992507935, "rewards/meter/std": 0.0004942560917697847, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9961405992507935, "rewards/total_composite/std": 0.0004942560917697847, "reward": 0.9961405992507935, "reward_std": 0.0004942500381730497, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006679588928818703, "sampling/sampling_logp_difference/max": 1.3542317152023315, "sampling/importance_sampling_ratio/min": 0.2581455409526825, "sampling/importance_sampling_ratio/mean": 0.9985120892524719, "sampling/importance_sampling_ratio/max": 1.1960804462432861, "entropy": 0.018420133623294532, "clip_ratio/low_mean": 0.0022727272007614374, "clip_ratio/low_min": 0.0022727272007614374, "clip_ratio/high_mean": 0.004545454401522875, "clip_ratio/high_max": 0.004545454401522875, "clip_ratio/region_mean": 0.006818181602284312, "reward_total_mean": 0.9961405992507935, "reward_meter_mean": 0.9961405992507935, "reward_meter_std": 0.0004942560917697847, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9961405992507935, "reward_total_composite_std": 0.0004942560917697847} {"timestamp_utc": "2026-04-12T01:55:10Z", "mode": "train", "global_step": 2353, "epoch": 0.09450937863999678, "loss": -0.0155, "grad_norm": 1.3235185146331787, "learning_rate": 2.872727272727273e-06, "num_tokens": 5329206.0, "completions/mean_length": 421.375, "completions/min_length": 398.0, "completions/max_length": 455.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 421.375, "completions/min_terminated_length": 398.0, "completions/max_terminated_length": 455.0, "rewards/meter/mean": 0.9977288246154785, "rewards/meter/std": 0.0008687536465004086, "rewards/count_adherence/mean": 0.6470588445663452, "rewards/count_adherence/std": 0.03144249692559242, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8467603325843811, "rewards/repeat_penalty/std": 0.08341042697429657, "rewards/total_composite/mean": 0.5467392206192017, "rewards/total_composite/std": 0.06185723468661308, "reward": 0.5467392206192017, "reward_std": 0.061857227236032486, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02322297915816307, "sampling/sampling_logp_difference/max": 1.9221699237823486, "sampling/importance_sampling_ratio/min": 0.1462891846895218, "sampling/importance_sampling_ratio/mean": 1.0037055015563965, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.19907046295702457, "clip_ratio/low_mean": 0.01009113050531596, "clip_ratio/low_min": 0.01009113050531596, "clip_ratio/high_mean": 0.007577877026051283, "clip_ratio/high_max": 0.007577877026051283, "clip_ratio/region_mean": 0.017669007531367242, "reward_total_mean": 0.5467392206192017, "reward_meter_mean": 0.9977288246154785, "reward_meter_std": 0.0008687536465004086, "reward_count_adherence_mean": 0.6470588445663452, "reward_count_adherence_std": 0.03144249692559242, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8467603325843811, "reward_repeat_penalty_std": 0.08341042697429657, "reward_total_composite_mean": 0.5467392206192017, "reward_total_composite_std": 0.06185723468661308} {"timestamp_utc": "2026-04-12T01:55:15Z", "mode": "train", "global_step": 2354, "epoch": 0.09454954412178174, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.86969696969697e-06, "num_tokens": 5330886.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0003920203307643533, "sampling/sampling_logp_difference/max": 0.0068152910098433495, "sampling/importance_sampling_ratio/min": 1.0, "sampling/importance_sampling_ratio/mean": 1.0003925561904907, "sampling/importance_sampling_ratio/max": 1.0068386793136597, "entropy": 0.0035217964032199234, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:55:23Z", "mode": "train", "global_step": 2355, "epoch": 0.09458970960356669, "loss": 0.0071, "grad_norm": 2.712887763977051, "learning_rate": 2.866666666666667e-06, "num_tokens": 5333484.0, "completions/mean_length": 168.75, "completions/min_length": 165.0, "completions/max_length": 171.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 168.75, "completions/min_terminated_length": 165.0, "completions/max_terminated_length": 171.0, "rewards/meter/mean": 0.9781259298324585, "rewards/meter/std": 0.03459608182311058, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8611111640930176, "rewards/repeat_penalty/std": 0.07856741547584534, "rewards/total_composite/mean": 0.8432746529579163, "rewards/total_composite/std": 0.09240440279245377, "reward": 0.8432746529579163, "reward_std": 0.09240438789129257, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027000876143574715, "sampling/sampling_logp_difference/max": 1.2758288383483887, "sampling/importance_sampling_ratio/min": 0.2791994512081146, "sampling/importance_sampling_ratio/mean": 1.0031359195709229, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1752516869455576, "clip_ratio/low_mean": 0.008841085364110768, "clip_ratio/low_min": 0.008841085364110768, "clip_ratio/high_mean": 0.017101168166846037, "clip_ratio/high_max": 0.017101168166846037, "clip_ratio/region_mean": 0.025942253530956805, "reward_total_mean": 0.8432746529579163, "reward_meter_mean": 0.9781259298324585, "reward_meter_std": 0.03459608182311058, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8611111640930176, "reward_repeat_penalty_std": 0.07856741547584534, "reward_total_composite_mean": 0.8432746529579163, "reward_total_composite_std": 0.09240440279245377} {"timestamp_utc": "2026-04-12T01:55:28Z", "mode": "train", "global_step": 2356, "epoch": 0.09462987508535164, "loss": 0.034, "grad_norm": 12.347071647644043, "learning_rate": 2.863636363636364e-06, "num_tokens": 5335199.0, "completions/mean_length": 44.375, "completions/min_length": 38.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.375, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.9444069266319275, "rewards/meter/std": 0.008035099133849144, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9444069266319275, "rewards/total_composite/std": 0.008035099133849144, "reward": 0.9444069266319275, "reward_std": 0.00803509633988142, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07534459978342056, "sampling/sampling_logp_difference/max": 2.617283821105957, "sampling/importance_sampling_ratio/min": 0.3516010642051697, "sampling/importance_sampling_ratio/mean": 1.0038548707962036, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.32474897243082523, "clip_ratio/low_mean": 0.030935976887121797, "clip_ratio/low_min": 0.030935976887121797, "clip_ratio/high_mean": 0.01955222897231579, "clip_ratio/high_max": 0.01955222897231579, "clip_ratio/region_mean": 0.050488205859437585, "reward_total_mean": 0.9444069266319275, "reward_meter_mean": 0.9444069266319275, "reward_meter_std": 0.008035099133849144, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9444069266319275, "reward_total_composite_std": 0.008035099133849144} {"timestamp_utc": "2026-04-12T01:55:33Z", "mode": "train", "global_step": 2357, "epoch": 0.0946700405671366, "loss": -0.0003, "grad_norm": 2.1621651649475098, "learning_rate": 2.860606060606061e-06, "num_tokens": 5337204.0, "completions/mean_length": 78.625, "completions/min_length": 77.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.625, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9971969127655029, "rewards/meter/std": 0.0005990342469885945, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9971969127655029, "rewards/total_composite/std": 0.0005990342469885945, "reward": 0.9971969127655029, "reward_std": 0.0005990189965814352, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021038906648755074, "sampling/sampling_logp_difference/max": 1.367281436920166, "sampling/importance_sampling_ratio/min": 0.2547987103462219, "sampling/importance_sampling_ratio/mean": 0.9987841844558716, "sampling/importance_sampling_ratio/max": 1.4412025213241577, "entropy": 0.12107675615698099, "clip_ratio/low_mean": 0.00634939968585968, "clip_ratio/low_min": 0.00634939968585968, "clip_ratio/high_mean": 0.009516005055047572, "clip_ratio/high_max": 0.009516005055047572, "clip_ratio/region_mean": 0.015865404740907252, "reward_total_mean": 0.9971969127655029, "reward_meter_mean": 0.9971969127655029, "reward_meter_std": 0.0005990342469885945, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9971969127655029, "reward_total_composite_std": 0.0005990342469885945} {"timestamp_utc": "2026-04-12T01:55:40Z", "mode": "train", "global_step": 2358, "epoch": 0.09471020604892155, "loss": -0.0096, "grad_norm": 3.5978167057037354, "learning_rate": 2.857575757575758e-06, "num_tokens": 5339049.0, "completions/mean_length": 73.625, "completions/min_length": 71.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.625, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9990395307540894, "rewards/meter/std": 0.00045778730418533087, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990395307540894, "rewards/total_composite/std": 0.00045778730418533087, "reward": 0.9990395307540894, "reward_std": 0.0004578051157295704, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0385035015642643, "sampling/sampling_logp_difference/max": 1.5631723403930664, "sampling/importance_sampling_ratio/min": 0.2094704955816269, "sampling/importance_sampling_ratio/mean": 0.999204158782959, "sampling/importance_sampling_ratio/max": 1.805698275566101, "entropy": 0.25467516854405403, "clip_ratio/low_mean": 0.02065259451046586, "clip_ratio/low_min": 0.02065259451046586, "clip_ratio/high_mean": 0.008355856058187783, "clip_ratio/high_max": 0.008355856058187783, "clip_ratio/region_mean": 0.029008450568653643, "reward_total_mean": 0.9990395307540894, "reward_meter_mean": 0.9990395307540894, "reward_meter_std": 0.00045778730418533087, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990395307540894, "reward_total_composite_std": 0.00045778730418533087} {"timestamp_utc": "2026-04-12T01:55:45Z", "mode": "train", "global_step": 2359, "epoch": 0.0947503715307065, "loss": 0.015, "grad_norm": 6.299208164215088, "learning_rate": 2.8545454545454548e-06, "num_tokens": 5340832.0, "completions/mean_length": 66.875, "completions/min_length": 65.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.875, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9785959720611572, "rewards/meter/std": 0.018975911661982536, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9785959720611572, "rewards/total_composite/std": 0.018975911661982536, "reward": 0.9785959720611572, "reward_std": 0.018975909799337387, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0360257662832737, "sampling/sampling_logp_difference/max": 1.296480655670166, "sampling/importance_sampling_ratio/min": 0.27349260449409485, "sampling/importance_sampling_ratio/mean": 1.0064629316329956, "sampling/importance_sampling_ratio/max": 1.6261426210403442, "entropy": 0.2618895582854748, "clip_ratio/low_mean": 0.007326300255954266, "clip_ratio/low_min": 0.007326300255954266, "clip_ratio/high_mean": 0.016904115793295205, "clip_ratio/high_max": 0.016904115793295205, "clip_ratio/region_mean": 0.02423041604924947, "reward_total_mean": 0.9785959720611572, "reward_meter_mean": 0.9785959720611572, "reward_meter_std": 0.018975911661982536, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9785959720611572, "reward_total_composite_std": 0.018975911661982536} {"timestamp_utc": "2026-04-12T01:55:50Z", "mode": "train", "global_step": 2360, "epoch": 0.09479053701249146, "loss": 0.0207, "grad_norm": 5.956533432006836, "learning_rate": 2.8515151515151516e-06, "num_tokens": 5342577.0, "completions/mean_length": 67.125, "completions/min_length": 65.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.125, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9699335098266602, "rewards/meter/std": 0.04688005894422531, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9699335098266602, "rewards/total_composite/std": 0.04688005894422531, "reward": 0.9699335098266602, "reward_std": 0.04688006266951561, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03932439908385277, "sampling/sampling_logp_difference/max": 1.0124170780181885, "sampling/importance_sampling_ratio/min": 0.36333972215652466, "sampling/importance_sampling_ratio/mean": 1.0127477645874023, "sampling/importance_sampling_ratio/max": 1.829169750213623, "entropy": 0.30276480689644814, "clip_ratio/low_mean": 0.0017857142956927419, "clip_ratio/low_min": 0.0017857142956927419, "clip_ratio/high_mean": 0.03938204434234649, "clip_ratio/high_max": 0.03938204434234649, "clip_ratio/region_mean": 0.04116775863803923, "reward_total_mean": 0.9699335098266602, "reward_meter_mean": 0.9699335098266602, "reward_meter_std": 0.04688005894422531, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9699335098266602, "reward_total_composite_std": 0.04688005894422531} {"timestamp_utc": "2026-04-12T01:55:57Z", "mode": "train", "global_step": 2361, "epoch": 0.09483070249427641, "loss": 0.0078, "grad_norm": 1.9891157150268555, "learning_rate": 2.848484848484849e-06, "num_tokens": 5345741.0, "completions/mean_length": 214.5, "completions/min_length": 210.0, "completions/max_length": 219.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 214.5, "completions/min_terminated_length": 210.0, "completions/max_terminated_length": 219.0, "rewards/meter/mean": 0.999156653881073, "rewards/meter/std": 0.00014262554759625345, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8409091234207153, "rewards/repeat_penalty/std": 0.10590588301420212, "rewards/total_composite/mean": 0.8401910066604614, "rewards/total_composite/std": 0.10572723299264908, "reward": 0.8401910066604614, "reward_std": 0.10572723299264908, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021222621202468872, "sampling/sampling_logp_difference/max": 1.204646110534668, "sampling/importance_sampling_ratio/min": 0.29979807138442993, "sampling/importance_sampling_ratio/mean": 1.002423882484436, "sampling/importance_sampling_ratio/max": 1.6807578802108765, "entropy": 0.19607393071055412, "clip_ratio/low_mean": 0.009289931389503181, "clip_ratio/low_min": 0.009289931389503181, "clip_ratio/high_mean": 0.00470314035192132, "clip_ratio/high_max": 0.00470314035192132, "clip_ratio/region_mean": 0.013993071741424501, "reward_total_mean": 0.8401910066604614, "reward_meter_mean": 0.999156653881073, "reward_meter_std": 0.00014262554759625345, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8409091234207153, "reward_repeat_penalty_std": 0.10590588301420212, "reward_total_composite_mean": 0.8401910066604614, "reward_total_composite_std": 0.10572723299264908} {"timestamp_utc": "2026-04-12T01:56:02Z", "mode": "train", "global_step": 2362, "epoch": 0.09487086797606137, "loss": -0.0082, "grad_norm": 3.921265125274658, "learning_rate": 2.8454545454545457e-06, "num_tokens": 5347168.0, "completions/mean_length": 40.375, "completions/min_length": 39.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.375, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.9963768124580383, "rewards/meter/std": 0.0008168795611709356, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9963768124580383, "rewards/total_composite/std": 0.0008168795611709356, "reward": 0.9963768124580383, "reward_std": 0.0008168675121851265, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018412932753562927, "sampling/sampling_logp_difference/max": 0.573250412940979, "sampling/importance_sampling_ratio/min": 0.5636902451515198, "sampling/importance_sampling_ratio/mean": 0.9972823858261108, "sampling/importance_sampling_ratio/max": 1.4037843942642212, "entropy": 0.08906339993700385, "clip_ratio/low_mean": 0.0032051282469183207, "clip_ratio/low_min": 0.0032051282469183207, "clip_ratio/high_mean": 0.006097560748457909, "clip_ratio/high_max": 0.006097560748457909, "clip_ratio/region_mean": 0.00930268899537623, "reward_total_mean": 0.9963768124580383, "reward_meter_mean": 0.9963768124580383, "reward_meter_std": 0.0008168795611709356, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9963768124580383, "reward_total_composite_std": 0.0008168795611709356} {"timestamp_utc": "2026-04-12T01:56:07Z", "mode": "train", "global_step": 2363, "epoch": 0.09491103345784632, "loss": -0.0237, "grad_norm": 4.108429908752441, "learning_rate": 2.8424242424242425e-06, "num_tokens": 5348939.0, "completions/mean_length": 74.375, "completions/min_length": 70.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.375, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9990479946136475, "rewards/meter/std": 0.0008222111500799656, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990479946136475, "rewards/total_composite/std": 0.0008222111500799656, "reward": 0.9990479946136475, "reward_std": 0.0008222072501666844, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03708229959011078, "sampling/sampling_logp_difference/max": 0.9699416160583496, "sampling/importance_sampling_ratio/min": 0.3791051506996155, "sampling/importance_sampling_ratio/mean": 0.9991539120674133, "sampling/importance_sampling_ratio/max": 1.825773000717163, "entropy": 0.2678355220705271, "clip_ratio/low_mean": 0.00890342053025961, "clip_ratio/low_min": 0.00890342053025961, "clip_ratio/high_mean": 0.021532694692723453, "clip_ratio/high_max": 0.021532694692723453, "clip_ratio/region_mean": 0.030436115222983062, "reward_total_mean": 0.9990479946136475, "reward_meter_mean": 0.9990479946136475, "reward_meter_std": 0.0008222111500799656, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990479946136475, "reward_total_composite_std": 0.0008222111500799656} {"timestamp_utc": "2026-04-12T01:56:13Z", "mode": "train", "global_step": 2364, "epoch": 0.09495119893963128, "loss": 0.0027, "grad_norm": 1.7912240028381348, "learning_rate": 2.83939393939394e-06, "num_tokens": 5351730.0, "completions/mean_length": 138.875, "completions/min_length": 138.0, "completions/max_length": 139.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 138.875, "completions/min_terminated_length": 138.0, "completions/max_terminated_length": 139.0, "rewards/meter/mean": 0.9963522553443909, "rewards/meter/std": 0.0003084685595240444, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8333333730697632, "rewards/repeat_penalty/std": 0.059391383081674576, "rewards/total_composite/mean": 0.8302946090698242, "rewards/total_composite/std": 0.0591915100812912, "reward": 0.8302946090698242, "reward_std": 0.05919152498245239, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015195811167359352, "sampling/sampling_logp_difference/max": 1.2972681522369385, "sampling/importance_sampling_ratio/min": 0.2732773423194885, "sampling/importance_sampling_ratio/mean": 0.9980447292327881, "sampling/importance_sampling_ratio/max": 1.877577304840088, "entropy": 0.07086402643471956, "clip_ratio/low_mean": 0.0054152331431396306, "clip_ratio/low_min": 0.0054152331431396306, "clip_ratio/high_mean": 0.01258992834482342, "clip_ratio/high_max": 0.01258992834482342, "clip_ratio/region_mean": 0.01800516148796305, "reward_total_mean": 0.8302946090698242, "reward_meter_mean": 0.9963522553443909, "reward_meter_std": 0.0003084685595240444, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8333333730697632, "reward_repeat_penalty_std": 0.059391383081674576, "reward_total_composite_mean": 0.8302946090698242, "reward_total_composite_std": 0.0591915100812912} {"timestamp_utc": "2026-04-12T01:56:18Z", "mode": "train", "global_step": 2365, "epoch": 0.09499136442141623, "loss": -0.0049, "grad_norm": 0.9692533612251282, "learning_rate": 2.8363636363636366e-06, "num_tokens": 5353652.0, "completions/mean_length": 73.25, "completions/min_length": 72.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.25, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9992170333862305, "rewards/meter/std": 0.00020559101540129632, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992170333862305, "rewards/total_composite/std": 0.00020559101540129632, "reward": 0.9992170333862305, "reward_std": 0.00020559433323796839, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023841191083192825, "sampling/sampling_logp_difference/max": 1.589078426361084, "sampling/importance_sampling_ratio/min": 0.20411363244056702, "sampling/importance_sampling_ratio/mean": 1.0054048299789429, "sampling/importance_sampling_ratio/max": 1.8762344121932983, "entropy": 0.20148388668894768, "clip_ratio/low_mean": 0.010369744850322604, "clip_ratio/low_min": 0.010369744850322604, "clip_ratio/high_mean": 0.01530504459515214, "clip_ratio/high_max": 0.01530504459515214, "clip_ratio/region_mean": 0.025674789445474744, "reward_total_mean": 0.9992170333862305, "reward_meter_mean": 0.9992170333862305, "reward_meter_std": 0.00020559101540129632, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992170333862305, "reward_total_composite_std": 0.00020559101540129632} {"timestamp_utc": "2026-04-12T01:56:25Z", "mode": "train", "global_step": 2366, "epoch": 0.09503152990320118, "loss": 0.0089, "grad_norm": 4.017276763916016, "learning_rate": 2.8333333333333335e-06, "num_tokens": 5356221.0, "completions/mean_length": 132.125, "completions/min_length": 129.0, "completions/max_length": 138.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.125, "completions/min_terminated_length": 129.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.9979519844055176, "rewards/meter/std": 0.0010567301651462913, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.9623122215270996, "rewards/total_composite/std": 0.06602418422698975, "reward": 0.9623122215270996, "reward_std": 0.06602419167757034, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05144790932536125, "sampling/sampling_logp_difference/max": 10.138971328735352, "sampling/importance_sampling_ratio/min": 3.95094248233363e-05, "sampling/importance_sampling_ratio/mean": 1.0044370889663696, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3485673386603594, "clip_ratio/low_mean": 0.01210981048643589, "clip_ratio/low_min": 0.01210981048643589, "clip_ratio/high_mean": 0.03128110943362117, "clip_ratio/high_max": 0.03128110943362117, "clip_ratio/region_mean": 0.04339091992005706, "reward_total_mean": 0.9623122215270996, "reward_meter_mean": 0.9979519844055176, "reward_meter_std": 0.0010567301651462913, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.9623122215270996, "reward_total_composite_std": 0.06602418422698975} {"timestamp_utc": "2026-04-12T01:56:31Z", "mode": "train", "global_step": 2367, "epoch": 0.09507169538498614, "loss": 0.0164, "grad_norm": 1.9662439823150635, "learning_rate": 2.8303030303030303e-06, "num_tokens": 5359095.0, "completions/mean_length": 154.25, "completions/min_length": 147.0, "completions/max_length": 160.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 154.25, "completions/min_terminated_length": 147.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.9776645302772522, "rewards/meter/std": 0.028479592874646187, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.930555522441864, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.9088918566703796, "rewards/total_composite/std": 0.04527009651064873, "reward": 0.9088918566703796, "reward_std": 0.04527008906006813, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023603463545441628, "sampling/sampling_logp_difference/max": 1.6123571395874023, "sampling/importance_sampling_ratio/min": 0.19941700994968414, "sampling/importance_sampling_ratio/mean": 1.0010682344436646, "sampling/importance_sampling_ratio/max": 1.7584105730056763, "entropy": 0.16689980775117874, "clip_ratio/low_mean": 0.007995150808710605, "clip_ratio/low_min": 0.007995150808710605, "clip_ratio/high_mean": 0.009202249813824892, "clip_ratio/high_max": 0.009202249813824892, "clip_ratio/region_mean": 0.017197400622535497, "reward_total_mean": 0.9088918566703796, "reward_meter_mean": 0.9776645302772522, "reward_meter_std": 0.028479592874646187, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.930555522441864, "reward_repeat_penalty_std": 0.05750546231865883, "reward_total_composite_mean": 0.9088918566703796, "reward_total_composite_std": 0.04527009651064873} {"timestamp_utc": "2026-04-12T01:56:37Z", "mode": "train", "global_step": 2368, "epoch": 0.09511186086677109, "loss": -0.005, "grad_norm": 3.341148853302002, "learning_rate": 2.8272727272727275e-06, "num_tokens": 5360933.0, "completions/mean_length": 67.75, "completions/min_length": 66.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.75, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.998802125453949, "rewards/meter/std": 0.00023434955801349133, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998802125453949, "rewards/total_composite/std": 0.00023434955801349133, "reward": 0.998802125453949, "reward_std": 0.00023435504408553243, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.036128342151641846, "sampling/sampling_logp_difference/max": 0.9030179977416992, "sampling/importance_sampling_ratio/min": 0.40534451603889465, "sampling/importance_sampling_ratio/mean": 1.0024096965789795, "sampling/importance_sampling_ratio/max": 1.526517629623413, "entropy": 0.2899752501398325, "clip_ratio/low_mean": 0.01125139684882015, "clip_ratio/low_min": 0.01125139684882015, "clip_ratio/high_mean": 0.02565775695256889, "clip_ratio/high_max": 0.02565775695256889, "clip_ratio/region_mean": 0.03690915380138904, "reward_total_mean": 0.998802125453949, "reward_meter_mean": 0.998802125453949, "reward_meter_std": 0.00023434955801349133, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.998802125453949, "reward_total_composite_std": 0.00023434955801349133} {"timestamp_utc": "2026-04-12T01:56:42Z", "mode": "train", "global_step": 2369, "epoch": 0.09515202634855605, "loss": 0.0001, "grad_norm": 1.1244981288909912, "learning_rate": 2.8242424242424244e-06, "num_tokens": 5362693.0, "completions/mean_length": 67.0, "completions/min_length": 67.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9980802536010742, "rewards/meter/std": 5.821814920636825e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980802536010742, "rewards/total_composite/std": 5.821814920636825e-05, "reward": 0.9980802536010742, "reward_std": 5.822332968818955e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011381209827959538, "sampling/sampling_logp_difference/max": 0.39493346214294434, "sampling/importance_sampling_ratio/min": 0.673724889755249, "sampling/importance_sampling_ratio/mean": 1.003217101097107, "sampling/importance_sampling_ratio/max": 1.3787553310394287, "entropy": 0.09162188414484262, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/high_mean": 0.0037313431967049837, "clip_ratio/high_max": 0.0037313431967049837, "clip_ratio/region_mean": 0.005597014795057476, "reward_total_mean": 0.9980802536010742, "reward_meter_mean": 0.9980802536010742, "reward_meter_std": 5.821814920636825e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9980802536010742, "reward_total_composite_std": 5.821814920636825e-05} {"timestamp_utc": "2026-04-12T01:56:46Z", "mode": "train", "global_step": 2370, "epoch": 0.095192191830341, "loss": 0.0382, "grad_norm": 12.456171035766602, "learning_rate": 2.821212121212121e-06, "num_tokens": 5364201.0, "completions/mean_length": 34.5, "completions/min_length": 34.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.5, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.9742438793182373, "rewards/meter/std": 0.03374186530709267, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9742438793182373, "rewards/total_composite/std": 0.03374186530709267, "reward": 0.9742438793182373, "reward_std": 0.03374185785651207, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02670416608452797, "sampling/sampling_logp_difference/max": 1.062800407409668, "sampling/importance_sampling_ratio/min": 0.34548693895339966, "sampling/importance_sampling_ratio/mean": 1.0038917064666748, "sampling/importance_sampling_ratio/max": 1.4533801078796387, "entropy": 0.21421233005821705, "clip_ratio/low_mean": 0.013544891262426972, "clip_ratio/low_min": 0.013544891262426972, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.013544891262426972, "reward_total_mean": 0.9742438793182373, "reward_meter_mean": 0.9742438793182373, "reward_meter_std": 0.03374186530709267, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9742438793182373, "reward_total_composite_std": 0.03374186530709267} {"timestamp_utc": "2026-04-12T01:56:51Z", "mode": "train", "global_step": 2371, "epoch": 0.09523235731212595, "loss": 0.0014, "grad_norm": 1.3944817781448364, "learning_rate": 2.818181818181818e-06, "num_tokens": 5366225.0, "completions/mean_length": 83.0, "completions/min_length": 83.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.0, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.9963405728340149, "rewards/meter/std": 0.00012567358498927206, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9963405728340149, "rewards/total_composite/std": 0.00012567358498927206, "reward": 0.9963405728340149, "reward_std": 0.00012567655357997864, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006072253920137882, "sampling/sampling_logp_difference/max": 1.5704057216644287, "sampling/importance_sampling_ratio/min": 0.20796078443527222, "sampling/importance_sampling_ratio/mean": 1.0000357627868652, "sampling/importance_sampling_ratio/max": 1.3552405834197998, "entropy": 0.030938314041122794, "clip_ratio/low_mean": 0.0015060240402817726, "clip_ratio/low_min": 0.0015060240402817726, "clip_ratio/high_mean": 0.0030120480805635452, "clip_ratio/high_max": 0.0030120480805635452, "clip_ratio/region_mean": 0.004518072120845318, "reward_total_mean": 0.9963405728340149, "reward_meter_mean": 0.9963405728340149, "reward_meter_std": 0.00012567358498927206, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9963405728340149, "reward_total_composite_std": 0.00012567358498927206} {"timestamp_utc": "2026-04-12T01:56:56Z", "mode": "train", "global_step": 2372, "epoch": 0.09527252279391091, "loss": 0.0096, "grad_norm": 3.56820011138916, "learning_rate": 2.8151515151515153e-06, "num_tokens": 5368757.0, "completions/mean_length": 132.5, "completions/min_length": 128.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.5, "completions/min_terminated_length": 128.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.9986804723739624, "rewards/meter/std": 0.0002552097721491009, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9808491468429565, "rewards/total_composite/std": 0.0504879355430603, "reward": 0.9808491468429565, "reward_std": 0.050487954169511795, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04589386284351349, "sampling/sampling_logp_difference/max": 1.5482797622680664, "sampling/importance_sampling_ratio/min": 0.21261338889598846, "sampling/importance_sampling_ratio/mean": 1.0080647468566895, "sampling/importance_sampling_ratio/max": 1.8720965385437012, "entropy": 0.38280507549643517, "clip_ratio/low_mean": 0.0055555556900799274, "clip_ratio/low_min": 0.0055555556900799274, "clip_ratio/high_mean": 0.03797508799470961, "clip_ratio/high_max": 0.03797508799470961, "clip_ratio/region_mean": 0.04353064368478954, "reward_total_mean": 0.9808491468429565, "reward_meter_mean": 0.9986804723739624, "reward_meter_std": 0.0002552097721491009, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.9808491468429565, "reward_total_composite_std": 0.0504879355430603} {"timestamp_utc": "2026-04-12T01:57:01Z", "mode": "train", "global_step": 2373, "epoch": 0.09531268827569586, "loss": 0.0088, "grad_norm": 2.409925699234009, "learning_rate": 2.812121212121212e-06, "num_tokens": 5371394.0, "completions/mean_length": 143.625, "completions/min_length": 141.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 143.625, "completions/min_terminated_length": 141.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.9991071224212646, "rewards/meter/std": 0.00023018992214929312, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.9634268879890442, "rewards/total_composite/std": 0.06610530614852905, "reward": 0.9634268879890442, "reward_std": 0.06610530614852905, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022872699424624443, "sampling/sampling_logp_difference/max": 1.4115514755249023, "sampling/importance_sampling_ratio/min": 0.24376478791236877, "sampling/importance_sampling_ratio/mean": 1.0038185119628906, "sampling/importance_sampling_ratio/max": 1.691011667251587, "entropy": 0.2039720769971609, "clip_ratio/low_mean": 0.00603448273614049, "clip_ratio/low_min": 0.00603448273614049, "clip_ratio/high_mean": 0.02439746167510748, "clip_ratio/high_max": 0.02439746167510748, "clip_ratio/region_mean": 0.03043194441124797, "reward_total_mean": 0.9634268879890442, "reward_meter_mean": 0.9991071224212646, "reward_meter_std": 0.00023018992214929312, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.9634268879890442, "reward_total_composite_std": 0.06610530614852905} {"timestamp_utc": "2026-04-12T01:57:06Z", "mode": "train", "global_step": 2374, "epoch": 0.09535285375748082, "loss": 0.0065, "grad_norm": 1.9035519361495972, "learning_rate": 2.809090909090909e-06, "num_tokens": 5373190.0, "completions/mean_length": 73.5, "completions/min_length": 71.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.5, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9992323517799377, "rewards/meter/std": 0.00020660254813265055, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992323517799377, "rewards/total_composite/std": 0.00020660254813265055, "reward": 0.9992323517799377, "reward_std": 0.00020661571761593223, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031100856140255928, "sampling/sampling_logp_difference/max": 1.3491382598876953, "sampling/importance_sampling_ratio/min": 0.25946375727653503, "sampling/importance_sampling_ratio/mean": 1.0032356977462769, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.20485389418900013, "clip_ratio/low_mean": 0.011739226174540818, "clip_ratio/low_min": 0.011739226174540818, "clip_ratio/high_mean": 0.013676133356057107, "clip_ratio/high_max": 0.013676133356057107, "clip_ratio/region_mean": 0.025415359530597925, "reward_total_mean": 0.9992323517799377, "reward_meter_mean": 0.9992323517799377, "reward_meter_std": 0.00020660254813265055, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992323517799377, "reward_total_composite_std": 0.00020660254813265055} {"timestamp_utc": "2026-04-12T01:57:14Z", "mode": "train", "global_step": 2375, "epoch": 0.09539301923926577, "loss": 0.0053, "grad_norm": 1.6295384168624878, "learning_rate": 2.806060606060606e-06, "num_tokens": 5377395.0, "completions/mean_length": 319.625, "completions/min_length": 317.0, "completions/max_length": 322.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 319.625, "completions/min_terminated_length": 317.0, "completions/max_terminated_length": 322.0, "rewards/meter/mean": 0.9988389015197754, "rewards/meter/std": 0.0002653080737218261, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8529411554336548, "rewards/repeat_penalty/std": 0.03144249692559242, "rewards/total_composite/mean": 0.7667578458786011, "rewards/total_composite/std": 0.02833147719502449, "reward": 0.7667578458786011, "reward_std": 0.02833147533237934, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02020336128771305, "sampling/sampling_logp_difference/max": 1.3237390518188477, "sampling/importance_sampling_ratio/min": 0.26613834500312805, "sampling/importance_sampling_ratio/mean": 1.0060769319534302, "sampling/importance_sampling_ratio/max": 1.622670292854309, "entropy": 0.20801045559346676, "clip_ratio/low_mean": 0.00859655940439552, "clip_ratio/low_min": 0.00859655940439552, "clip_ratio/high_mean": 0.006286898045800626, "clip_ratio/high_max": 0.006286898045800626, "clip_ratio/region_mean": 0.014883457450196147, "reward_total_mean": 0.7667578458786011, "reward_meter_mean": 0.9988389015197754, "reward_meter_std": 0.0002653080737218261, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8529411554336548, "reward_repeat_penalty_std": 0.03144249692559242, "reward_total_composite_mean": 0.7667578458786011, "reward_total_composite_std": 0.02833147719502449} {"timestamp_utc": "2026-04-12T01:57:19Z", "mode": "train", "global_step": 2376, "epoch": 0.09543318472105072, "loss": 0.0009, "grad_norm": 0.18841330707073212, "learning_rate": 2.803030303030303e-06, "num_tokens": 5379314.0, "completions/mean_length": 66.875, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.875, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981135129928589, "rewards/meter/std": 1.4713088603457436e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981135129928589, "rewards/total_composite/std": 1.4713088603457436e-05, "reward": 0.9981135129928589, "reward_std": 1.4704202840221114e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010610595345497131, "sampling/sampling_logp_difference/max": 0.893486499786377, "sampling/importance_sampling_ratio/min": 0.40922650694847107, "sampling/importance_sampling_ratio/mean": 0.9993737936019897, "sampling/importance_sampling_ratio/max": 1.2641924619674683, "entropy": 0.0682340869680047, "clip_ratio/low_mean": 0.005597014795057476, "clip_ratio/low_min": 0.005597014795057476, "clip_ratio/high_mean": 0.007490954245440662, "clip_ratio/high_max": 0.007490954245440662, "clip_ratio/region_mean": 0.013087969040498137, "reward_total_mean": 0.9981135129928589, "reward_meter_mean": 0.9981135129928589, "reward_meter_std": 1.4713088603457436e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981135129928589, "reward_total_composite_std": 1.4713088603457436e-05} {"timestamp_utc": "2026-04-12T01:57:24Z", "mode": "train", "global_step": 2377, "epoch": 0.09547335020283568, "loss": -0.001, "grad_norm": 0.5763253569602966, "learning_rate": 2.8000000000000003e-06, "num_tokens": 5381154.0, "completions/mean_length": 67.0, "completions/min_length": 67.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981157779693604, "rewards/meter/std": 2.6794468794832937e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981157779693604, "rewards/total_composite/std": 2.6794468794832937e-05, "reward": 0.9981157779693604, "reward_std": 2.6794450604938902e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008252451196312904, "sampling/sampling_logp_difference/max": 0.4723522663116455, "sampling/importance_sampling_ratio/min": 0.6235338449478149, "sampling/importance_sampling_ratio/mean": 1.001448631286621, "sampling/importance_sampling_ratio/max": 1.3120979070663452, "entropy": 0.06305565731599927, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.005597014795057476, "clip_ratio/high_max": 0.005597014795057476, "clip_ratio/region_mean": 0.005597014795057476, "reward_total_mean": 0.9981157779693604, "reward_meter_mean": 0.9981157779693604, "reward_meter_std": 2.6794468794832937e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981157779693604, "reward_total_composite_std": 2.6794468794832937e-05} {"timestamp_utc": "2026-04-12T01:57:30Z", "mode": "train", "global_step": 2378, "epoch": 0.09551351568462063, "loss": 0.0034, "grad_norm": 1.9172272682189941, "learning_rate": 2.7969696969696976e-06, "num_tokens": 5384219.0, "completions/mean_length": 186.125, "completions/min_length": 184.0, "completions/max_length": 189.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 186.125, "completions/min_terminated_length": 184.0, "completions/max_terminated_length": 189.0, "rewards/meter/mean": 0.9751391410827637, "rewards/meter/std": 0.04846389219164848, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8181818723678589, "rewards/repeat_penalty/std": 0.08416548371315002, "rewards/total_composite/mean": 0.7978767156600952, "rewards/total_composite/std": 0.09277582913637161, "reward": 0.7978767156600952, "reward_std": 0.09277583658695221, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02777726948261261, "sampling/sampling_logp_difference/max": 0.9550704956054688, "sampling/importance_sampling_ratio/min": 0.3847850263118744, "sampling/importance_sampling_ratio/mean": 1.005282998085022, "sampling/importance_sampling_ratio/max": 1.715010404586792, "entropy": 0.2459548767656088, "clip_ratio/low_mean": 0.004003660404123366, "clip_ratio/low_min": 0.004003660404123366, "clip_ratio/high_mean": 0.015454705455340445, "clip_ratio/high_max": 0.015454705455340445, "clip_ratio/region_mean": 0.01945836585946381, "reward_total_mean": 0.7978767156600952, "reward_meter_mean": 0.9751391410827637, "reward_meter_std": 0.04846389219164848, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8181818723678589, "reward_repeat_penalty_std": 0.08416548371315002, "reward_total_composite_mean": 0.7978767156600952, "reward_total_composite_std": 0.09277582913637161} {"timestamp_utc": "2026-04-12T01:57:35Z", "mode": "train", "global_step": 2379, "epoch": 0.09555368116640559, "loss": 0.0033, "grad_norm": 2.5564944744110107, "learning_rate": 2.7939393939393944e-06, "num_tokens": 5385952.0, "completions/mean_length": 67.625, "completions/min_length": 65.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.625, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9991129636764526, "rewards/meter/std": 0.00021267922420520335, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991129636764526, "rewards/total_composite/std": 0.00021267922420520335, "reward": 0.9991129636764526, "reward_std": 0.00021267762349452823, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03177187591791153, "sampling/sampling_logp_difference/max": 0.6790390014648438, "sampling/importance_sampling_ratio/min": 0.5071040987968445, "sampling/importance_sampling_ratio/mean": 1.011904001235962, "sampling/importance_sampling_ratio/max": 1.5362749099731445, "entropy": 0.32505345717072487, "clip_ratio/low_mean": 0.01117487601004541, "clip_ratio/low_min": 0.01117487601004541, "clip_ratio/high_mean": 0.00916612590663135, "clip_ratio/high_max": 0.00916612590663135, "clip_ratio/region_mean": 0.02034100191667676, "reward_total_mean": 0.9991129636764526, "reward_meter_mean": 0.9991129636764526, "reward_meter_std": 0.00021267922420520335, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9991129636764526, "reward_total_composite_std": 0.00021267922420520335} {"timestamp_utc": "2026-04-12T01:57:40Z", "mode": "train", "global_step": 2380, "epoch": 0.09559384664819054, "loss": -0.0004, "grad_norm": 0.42718541622161865, "learning_rate": 2.7909090909090912e-06, "num_tokens": 5387928.0, "completions/mean_length": 67.0, "completions/min_length": 67.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981181621551514, "rewards/meter/std": 2.501497874618508e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981181621551514, "rewards/total_composite/std": 2.501497874618508e-05, "reward": 0.9981181621551514, "reward_std": 2.5023065973073244e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006671510171145201, "sampling/sampling_logp_difference/max": 1.0379719734191895, "sampling/importance_sampling_ratio/min": 0.3541722297668457, "sampling/importance_sampling_ratio/mean": 0.9986084699630737, "sampling/importance_sampling_ratio/max": 1.1040713787078857, "entropy": 0.024645814672112465, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/high_mean": 0.0018656715983524919, "clip_ratio/high_max": 0.0018656715983524919, "clip_ratio/region_mean": 0.0037313431967049837, "reward_total_mean": 0.9981181621551514, "reward_meter_mean": 0.9981181621551514, "reward_meter_std": 2.501497874618508e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981181621551514, "reward_total_composite_std": 2.501497874618508e-05} {"timestamp_utc": "2026-04-12T01:57:45Z", "mode": "train", "global_step": 2381, "epoch": 0.09563401212997549, "loss": 0.0134, "grad_norm": 3.057466745376587, "learning_rate": 2.7878787878787885e-06, "num_tokens": 5390258.0, "completions/mean_length": 107.25, "completions/min_length": 105.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.25, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.9992028474807739, "rewards/meter/std": 7.249469490488991e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9742231369018555, "rewards/total_composite/std": 0.07065970450639725, "reward": 0.9742231369018555, "reward_std": 0.07065972685813904, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020719556137919426, "sampling/sampling_logp_difference/max": 0.7676267623901367, "sampling/importance_sampling_ratio/min": 0.4641132354736328, "sampling/importance_sampling_ratio/mean": 1.0024255514144897, "sampling/importance_sampling_ratio/max": 1.8616714477539062, "entropy": 0.1429230272769928, "clip_ratio/low_mean": 0.0011261261533945799, "clip_ratio/low_min": 0.0011261261533945799, "clip_ratio/high_mean": 0.02113948343321681, "clip_ratio/high_max": 0.02113948343321681, "clip_ratio/region_mean": 0.02226560958661139, "reward_total_mean": 0.9742231369018555, "reward_meter_mean": 0.9992028474807739, "reward_meter_std": 7.249469490488991e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9742231369018555, "reward_total_composite_std": 0.07065970450639725} {"timestamp_utc": "2026-04-12T01:57:50Z", "mode": "train", "global_step": 2382, "epoch": 0.09567417761176045, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.7848484848484853e-06, "num_tokens": 5392402.0, "completions/mean_length": 80.0, "completions/min_length": 80.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.0, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "reward": 0.7190229296684265, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0010491692228242755, "sampling/sampling_logp_difference/max": 0.07737492024898529, "sampling/importance_sampling_ratio/min": 0.9255428314208984, "sampling/importance_sampling_ratio/mean": 1.0004812479019165, "sampling/importance_sampling_ratio/max": 1.0541869401931763, "entropy": 0.008781549287959933, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7190229296684265, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:57:56Z", "mode": "train", "global_step": 2383, "epoch": 0.0957143430935454, "loss": 0.006, "grad_norm": 1.7132611274719238, "learning_rate": 2.781818181818182e-06, "num_tokens": 5394915.0, "completions/mean_length": 130.125, "completions/min_length": 130.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 130.125, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.999283492565155, "rewards/meter/std": 2.7988780857413076e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9457492828369141, "rewards/total_composite/std": 0.07386652380228043, "reward": 0.9457492828369141, "reward_std": 0.07386651635169983, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016906656324863434, "sampling/sampling_logp_difference/max": 1.372336745262146, "sampling/importance_sampling_ratio/min": 0.2535138428211212, "sampling/importance_sampling_ratio/mean": 1.003693699836731, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09555556159466505, "clip_ratio/low_mean": 0.0076776278438046575, "clip_ratio/low_min": 0.0076776278438046575, "clip_ratio/high_mean": 0.014423077227547765, "clip_ratio/high_max": 0.014423077227547765, "clip_ratio/region_mean": 0.022100705071352422, "reward_total_mean": 0.9457492828369141, "reward_meter_mean": 0.999283492565155, "reward_meter_std": 2.7988780857413076e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.9457492828369141, "reward_total_composite_std": 0.07386652380228043} {"timestamp_utc": "2026-04-12T01:58:01Z", "mode": "train", "global_step": 2384, "epoch": 0.09575450857533035, "loss": 0.0197, "grad_norm": 4.2915167808532715, "learning_rate": 2.778787878787879e-06, "num_tokens": 5397220.0, "completions/mean_length": 101.125, "completions/min_length": 99.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.125, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.9984184503555298, "rewards/meter/std": 0.0006285551935434341, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9984184503555298, "rewards/total_composite/std": 0.0006285551935434341, "reward": 0.9984184503555298, "reward_std": 0.0006285490817390382, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05373314768075943, "sampling/sampling_logp_difference/max": 1.02024507522583, "sampling/importance_sampling_ratio/min": 0.3605065941810608, "sampling/importance_sampling_ratio/mean": 1.0017614364624023, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.41279827430844307, "clip_ratio/low_mean": 0.01336255669593811, "clip_ratio/low_min": 0.01336255669593811, "clip_ratio/high_mean": 0.03354567987844348, "clip_ratio/high_max": 0.03354567987844348, "clip_ratio/region_mean": 0.04690823657438159, "reward_total_mean": 0.9984184503555298, "reward_meter_mean": 0.9984184503555298, "reward_meter_std": 0.0006285551935434341, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9984184503555298, "reward_total_composite_std": 0.0006285551935434341} {"timestamp_utc": "2026-04-12T01:58:05Z", "mode": "train", "global_step": 2385, "epoch": 0.09579467405711531, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.7757575757575762e-06, "num_tokens": 5398716.0, "completions/mean_length": 35.0, "completions/min_length": 35.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9980231523513794, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980231523513794, "rewards/total_composite/std": 0.0, "reward": 0.9980231523513794, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0007748155039735138, "sampling/sampling_logp_difference/max": 0.023845985531806946, "sampling/importance_sampling_ratio/min": 0.9885799884796143, "sampling/importance_sampling_ratio/mean": 1.0006725788116455, "sampling/importance_sampling_ratio/max": 1.024132490158081, "entropy": 0.009345798986032605, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9980231523513794, "reward_meter_mean": 0.9980231523513794, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9980231523513794, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:58:09Z", "mode": "train", "global_step": 2386, "epoch": 0.09583483953890026, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.772727272727273e-06, "num_tokens": 5400332.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00041769127710722387, "sampling/sampling_logp_difference/max": 0.010340280830860138, "sampling/importance_sampling_ratio/min": 0.9897130131721497, "sampling/importance_sampling_ratio/mean": 1.0003135204315186, "sampling/importance_sampling_ratio/max": 1.0083444118499756, "entropy": 0.004995487950509414, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:58:14Z", "mode": "train", "global_step": 2387, "epoch": 0.09587500502068522, "loss": 0.0021, "grad_norm": 3.3865015506744385, "learning_rate": 2.76969696969697e-06, "num_tokens": 5402149.0, "completions/mean_length": 68.125, "completions/min_length": 65.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.125, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9987808465957642, "rewards/meter/std": 0.0002124947786796838, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9987808465957642, "rewards/total_composite/std": 0.0002124947786796838, "reward": 0.9987808465957642, "reward_std": 0.00021250976715236902, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05237334594130516, "sampling/sampling_logp_difference/max": 1.7108659744262695, "sampling/importance_sampling_ratio/min": 0.18070924282073975, "sampling/importance_sampling_ratio/mean": 1.0035470724105835, "sampling/importance_sampling_ratio/max": 1.5863909721374512, "entropy": 0.371359147131443, "clip_ratio/low_mean": 0.016549831489101052, "clip_ratio/low_min": 0.016549831489101052, "clip_ratio/high_mean": 0.019927536603063345, "clip_ratio/high_max": 0.019927536603063345, "clip_ratio/region_mean": 0.0364773680921644, "reward_total_mean": 0.9987808465957642, "reward_meter_mean": 0.9987808465957642, "reward_meter_std": 0.0002124947786796838, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9987808465957642, "reward_total_composite_std": 0.0002124947786796838} {"timestamp_utc": "2026-04-12T01:58:18Z", "mode": "train", "global_step": 2388, "epoch": 0.09591517050247017, "loss": 0.0005, "grad_norm": 2.705258846282959, "learning_rate": 2.766666666666667e-06, "num_tokens": 5403597.0, "completions/mean_length": 32.0, "completions/min_length": 32.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9992798566818237, "rewards/meter/std": 3.3992837416008115e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992798566818237, "rewards/total_composite/std": 3.3992837416008115e-05, "reward": 0.9992798566818237, "reward_std": 3.3992837416008115e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00878862477838993, "sampling/sampling_logp_difference/max": 1.0551605224609375, "sampling/importance_sampling_ratio/min": 0.3481365442276001, "sampling/importance_sampling_ratio/mean": 0.9955343008041382, "sampling/importance_sampling_ratio/max": 1.0776525735855103, "entropy": 0.015011629671789706, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.00390625, "clip_ratio/high_max": 0.00390625, "clip_ratio/region_mean": 0.00390625, "reward_total_mean": 0.9992798566818237, "reward_meter_mean": 0.9992798566818237, "reward_meter_std": 3.3992837416008115e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992798566818237, "reward_total_composite_std": 3.3992837416008115e-05} {"timestamp_utc": "2026-04-12T01:58:23Z", "mode": "train", "global_step": 2389, "epoch": 0.09595533598425512, "loss": 0.005, "grad_norm": 3.947572946548462, "learning_rate": 2.763636363636364e-06, "num_tokens": 5405494.0, "completions/mean_length": 68.125, "completions/min_length": 65.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.125, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9979345798492432, "rewards/meter/std": 0.0023531855549663305, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979345798492432, "rewards/total_composite/std": 0.0023531855549663305, "reward": 0.9979345798492432, "reward_std": 0.0023531648330390453, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03587198257446289, "sampling/sampling_logp_difference/max": 0.9495134353637695, "sampling/importance_sampling_ratio/min": 0.3869292438030243, "sampling/importance_sampling_ratio/mean": 1.0087910890579224, "sampling/importance_sampling_ratio/max": 1.5146595239639282, "entropy": 0.3192524388432503, "clip_ratio/low_mean": 0.0071428571827709675, "clip_ratio/low_min": 0.0071428571827709675, "clip_ratio/high_mean": 0.027370785945095122, "clip_ratio/high_max": 0.027370785945095122, "clip_ratio/region_mean": 0.03451364312786609, "reward_total_mean": 0.9979345798492432, "reward_meter_mean": 0.9979345798492432, "reward_meter_std": 0.0023531855549663305, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9979345798492432, "reward_total_composite_std": 0.0023531855549663305} {"timestamp_utc": "2026-04-12T01:58:28Z", "mode": "train", "global_step": 2390, "epoch": 0.09599550146604008, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.760606060606061e-06, "num_tokens": 5407279.0, "completions/mean_length": 54.125, "completions/min_length": 54.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.125, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0010611445177346468, "sampling/sampling_logp_difference/max": 0.17297124862670898, "sampling/importance_sampling_ratio/min": 0.841161847114563, "sampling/importance_sampling_ratio/mean": 0.9997290968894958, "sampling/importance_sampling_ratio/max": 1.0198215246200562, "entropy": 0.004519450158113614, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T01:58:33Z", "mode": "train", "global_step": 2391, "epoch": 0.09603566694782503, "loss": -0.0026, "grad_norm": 1.7748810052871704, "learning_rate": 2.7575757575757576e-06, "num_tokens": 5409667.0, "completions/mean_length": 121.5, "completions/min_length": 121.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 121.5, "completions/min_terminated_length": 121.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.9796151518821716, "rewards/meter/std": 0.002762306947261095, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9271951913833618, "rewards/total_composite/std": 0.0733107179403305, "reward": 0.9271951913833618, "reward_std": 0.07331071048974991, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008518277667462826, "sampling/sampling_logp_difference/max": 1.3747358322143555, "sampling/importance_sampling_ratio/min": 0.2529063820838928, "sampling/importance_sampling_ratio/mean": 1.0016850233078003, "sampling/importance_sampling_ratio/max": 1.2646301984786987, "entropy": 0.05755401449277997, "clip_ratio/low_mean": 0.001033057807944715, "clip_ratio/low_min": 0.001033057807944715, "clip_ratio/high_mean": 0.005099173518829048, "clip_ratio/high_max": 0.005099173518829048, "clip_ratio/region_mean": 0.006132231326773763, "reward_total_mean": 0.9271951913833618, "reward_meter_mean": 0.9796151518821716, "reward_meter_std": 0.002762306947261095, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.9271951913833618, "reward_total_composite_std": 0.0733107179403305} {"timestamp_utc": "2026-04-12T01:58:39Z", "mode": "train", "global_step": 2392, "epoch": 0.09607583242960999, "loss": -0.002, "grad_norm": 2.4178664684295654, "learning_rate": 2.754545454545455e-06, "num_tokens": 5412340.0, "completions/mean_length": 154.125, "completions/min_length": 151.0, "completions/max_length": 157.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 154.125, "completions/min_terminated_length": 151.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.9878800511360168, "rewards/meter/std": 0.02659641019999981, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.910714328289032, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9002017974853516, "rewards/total_composite/std": 0.08359287679195404, "reward": 0.9002017974853516, "reward_std": 0.08359287679195404, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03138583526015282, "sampling/sampling_logp_difference/max": 1.6040925979614258, "sampling/importance_sampling_ratio/min": 0.20107191801071167, "sampling/importance_sampling_ratio/mean": 1.0050873756408691, "sampling/importance_sampling_ratio/max": 1.3937476873397827, "entropy": 0.23401772044599056, "clip_ratio/low_mean": 0.011445078765973449, "clip_ratio/low_min": 0.011445078765973449, "clip_ratio/high_mean": 0.007223089050967246, "clip_ratio/high_max": 0.007223089050967246, "clip_ratio/region_mean": 0.018668167816940695, "reward_total_mean": 0.9002017974853516, "reward_meter_mean": 0.9878800511360168, "reward_meter_std": 0.02659641019999981, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.910714328289032, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.9002017974853516, "reward_total_composite_std": 0.08359287679195404} {"timestamp_utc": "2026-04-12T01:58:44Z", "mode": "train", "global_step": 2393, "epoch": 0.09611599791139494, "loss": 0.0047, "grad_norm": 1.324416995048523, "learning_rate": 2.7515151515151517e-06, "num_tokens": 5414319.0, "completions/mean_length": 72.375, "completions/min_length": 72.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.375, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9992770552635193, "rewards/meter/std": 0.00018226148677058518, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992770552635193, "rewards/total_composite/std": 0.00018226148677058518, "reward": 0.9992770552635193, "reward_std": 0.00018225403618998826, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020829232409596443, "sampling/sampling_logp_difference/max": 1.3297691345214844, "sampling/importance_sampling_ratio/min": 0.26453831791877747, "sampling/importance_sampling_ratio/mean": 1.0025677680969238, "sampling/importance_sampling_ratio/max": 1.7886258363723755, "entropy": 0.11769631877541542, "clip_ratio/low_mean": 0.006779896095395088, "clip_ratio/low_min": 0.006779896095395088, "clip_ratio/high_mean": 0.005208333372138441, "clip_ratio/high_max": 0.005208333372138441, "clip_ratio/region_mean": 0.011988229467533529, "reward_total_mean": 0.9992770552635193, "reward_meter_mean": 0.9992770552635193, "reward_meter_std": 0.00018226148677058518, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992770552635193, "reward_total_composite_std": 0.00018226148677058518} {"timestamp_utc": "2026-04-12T01:58:48Z", "mode": "train", "global_step": 2394, "epoch": 0.09615616339317991, "loss": 0.0004, "grad_norm": 1.3910056352615356, "learning_rate": 2.7484848484848486e-06, "num_tokens": 5416127.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993087649345398, "rewards/meter/std": 5.955179949523881e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993087649345398, "rewards/total_composite/std": 5.955179949523881e-05, "reward": 0.9993087649345398, "reward_std": 5.955949382041581e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012217044830322266, "sampling/sampling_logp_difference/max": 0.748956024646759, "sampling/importance_sampling_ratio/min": 0.4728599488735199, "sampling/importance_sampling_ratio/mean": 0.9990496635437012, "sampling/importance_sampling_ratio/max": 1.3544596433639526, "entropy": 0.047464726492762566, "clip_ratio/low_mean": 0.01171875, "clip_ratio/low_min": 0.01171875, "clip_ratio/high_mean": 0.0078125, "clip_ratio/high_max": 0.0078125, "clip_ratio/region_mean": 0.01953125, "reward_total_mean": 0.9993087649345398, "reward_meter_mean": 0.9993087649345398, "reward_meter_std": 5.955179949523881e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993087649345398, "reward_total_composite_std": 5.955179949523881e-05} {"timestamp_utc": "2026-04-12T01:58:53Z", "mode": "train", "global_step": 2395, "epoch": 0.09619632887496486, "loss": 0.0, "grad_norm": 0.06802152097225189, "learning_rate": 2.7454545454545454e-06, "num_tokens": 5417895.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993706345558167, "rewards/meter/std": 5.743691417592345e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993706345558167, "rewards/total_composite/std": 5.743691417592345e-06, "reward": 0.9993706345558167, "reward_std": 5.761006832472049e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008546598255634308, "sampling/sampling_logp_difference/max": 0.5136269330978394, "sampling/importance_sampling_ratio/min": 0.5983215570449829, "sampling/importance_sampling_ratio/mean": 1.002130389213562, "sampling/importance_sampling_ratio/max": 1.6340101957321167, "entropy": 0.038509635254740715, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0078125, "clip_ratio/high_max": 0.0078125, "clip_ratio/region_mean": 0.0078125, "reward_total_mean": 0.9993706345558167, "reward_meter_mean": 0.9993706345558167, "reward_meter_std": 5.743691417592345e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993706345558167, "reward_total_composite_std": 5.743691417592345e-06} {"timestamp_utc": "2026-04-12T01:58:58Z", "mode": "train", "global_step": 2396, "epoch": 0.09623649435674982, "loss": 0.0113, "grad_norm": 1.6420077085494995, "learning_rate": 2.7424242424242426e-06, "num_tokens": 5420082.0, "completions/mean_length": 102.375, "completions/min_length": 100.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.375, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.9934667944908142, "rewards/meter/std": 0.006176711525768042, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.968574583530426, "rewards/total_composite/std": 0.06977622210979462, "reward": 0.968574583530426, "reward_std": 0.06977621465921402, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022047532722353935, "sampling/sampling_logp_difference/max": 0.9480593204498291, "sampling/importance_sampling_ratio/min": 0.38789549469947815, "sampling/importance_sampling_ratio/mean": 1.0029125213623047, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.13659420423209667, "clip_ratio/low_mean": 0.0012019231216982007, "clip_ratio/low_min": 0.0012019231216982007, "clip_ratio/high_mean": 0.015861412626691163, "clip_ratio/high_max": 0.015861412626691163, "clip_ratio/region_mean": 0.017063335748389363, "reward_total_mean": 0.968574583530426, "reward_meter_mean": 0.9934667944908142, "reward_meter_std": 0.006176711525768042, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.968574583530426, "reward_total_composite_std": 0.06977622210979462} {"timestamp_utc": "2026-04-12T01:59:04Z", "mode": "train", "global_step": 2397, "epoch": 0.09627665983853477, "loss": -0.0055, "grad_norm": 11.099533081054688, "learning_rate": 2.7393939393939395e-06, "num_tokens": 5422174.0, "completions/mean_length": 84.5, "completions/min_length": 84.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.5, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.9455694556236267, "rewards/meter/std": 0.006192359142005444, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9455694556236267, "rewards/total_composite/std": 0.006192359142005444, "reward": 0.9455694556236267, "reward_std": 0.00619234936311841, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014308757148683071, "sampling/sampling_logp_difference/max": 0.9953222274780273, "sampling/importance_sampling_ratio/min": 0.36960431933403015, "sampling/importance_sampling_ratio/mean": 0.9979245662689209, "sampling/importance_sampling_ratio/max": 1.6855740547180176, "entropy": 0.06057103397324681, "clip_ratio/low_mean": 0.008893557591363788, "clip_ratio/low_min": 0.008893557591363788, "clip_ratio/high_mean": 0.0028735632076859474, "clip_ratio/high_max": 0.0028735632076859474, "clip_ratio/region_mean": 0.011767120799049735, "reward_total_mean": 0.9455694556236267, "reward_meter_mean": 0.9455694556236267, "reward_meter_std": 0.006192359142005444, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9455694556236267, "reward_total_composite_std": 0.006192359142005444} {"timestamp_utc": "2026-04-12T01:59:10Z", "mode": "train", "global_step": 2398, "epoch": 0.09631682532031972, "loss": -0.0017, "grad_norm": 1.1203114986419678, "learning_rate": 2.7363636363636363e-06, "num_tokens": 5424845.0, "completions/mean_length": 147.875, "completions/min_length": 144.0, "completions/max_length": 152.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 147.875, "completions/min_terminated_length": 144.0, "completions/max_terminated_length": 152.0, "rewards/meter/mean": 0.9743971824645996, "rewards/meter/std": 0.0041354927234351635, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.819444477558136, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.7985101938247681, "rewards/total_composite/std": 0.05690953880548477, "reward": 0.7985101938247681, "reward_std": 0.05690954998135567, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011587115004658699, "sampling/sampling_logp_difference/max": 1.6029415130615234, "sampling/importance_sampling_ratio/min": 0.20130351185798645, "sampling/importance_sampling_ratio/mean": 0.9987753629684448, "sampling/importance_sampling_ratio/max": 1.5559734106063843, "entropy": 0.04592153197154403, "clip_ratio/low_mean": 0.003401839407160878, "clip_ratio/low_min": 0.003401839407160878, "clip_ratio/high_mean": 0.004206492216326296, "clip_ratio/high_max": 0.004206492216326296, "clip_ratio/region_mean": 0.0076083316234871745, "reward_total_mean": 0.7985101938247681, "reward_meter_mean": 0.9743971824645996, "reward_meter_std": 0.0041354927234351635, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.819444477558136, "reward_repeat_penalty_std": 0.05750546231865883, "reward_total_composite_mean": 0.7985101938247681, "reward_total_composite_std": 0.05690953880548477} {"timestamp_utc": "2026-04-12T01:59:15Z", "mode": "train", "global_step": 2399, "epoch": 0.09635699080210468, "loss": 0.0016, "grad_norm": 1.154660701751709, "learning_rate": 2.7333333333333336e-06, "num_tokens": 5427116.0, "completions/mean_length": 96.875, "completions/min_length": 96.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.875, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.9992877244949341, "rewards/meter/std": 5.900250835111365e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992877244949341, "rewards/total_composite/std": 5.900250835111365e-05, "reward": 0.9992877244949341, "reward_std": 5.900421820115298e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012875314801931381, "sampling/sampling_logp_difference/max": 1.2882099151611328, "sampling/importance_sampling_ratio/min": 0.27576398849487305, "sampling/importance_sampling_ratio/mean": 0.9967014789581299, "sampling/importance_sampling_ratio/max": 1.3556400537490845, "entropy": 0.05716710351407528, "clip_ratio/low_mean": 0.0013020833721384406, "clip_ratio/low_min": 0.0013020833721384406, "clip_ratio/high_mean": 0.015463917283341289, "clip_ratio/high_max": 0.015463917283341289, "clip_ratio/region_mean": 0.01676600065547973, "reward_total_mean": 0.9992877244949341, "reward_meter_mean": 0.9992877244949341, "reward_meter_std": 5.900250835111365e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992877244949341, "reward_total_composite_std": 5.900250835111365e-05} {"timestamp_utc": "2026-04-12T01:59:20Z", "mode": "train", "global_step": 2400, "epoch": 0.09639715628388963, "loss": 0.0041, "grad_norm": 1.3548955917358398, "learning_rate": 2.7303030303030304e-06, "num_tokens": 5428869.0, "completions/mean_length": 72.125, "completions/min_length": 71.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.125, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9993604421615601, "rewards/meter/std": 9.307587606599554e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993604421615601, "rewards/total_composite/std": 9.307587606599554e-05, "reward": 0.9993604421615601, "reward_std": 9.308403969043866e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01761528290808201, "sampling/sampling_logp_difference/max": 1.1383390426635742, "sampling/importance_sampling_ratio/min": 0.32035067677497864, "sampling/importance_sampling_ratio/mean": 1.0040256977081299, "sampling/importance_sampling_ratio/max": 1.8390227556228638, "entropy": 0.07446162821725011, "clip_ratio/low_mean": 0.004999999888241291, "clip_ratio/low_min": 0.004999999888241291, "clip_ratio/high_mean": 0.0034966744715347886, "clip_ratio/high_max": 0.0034966744715347886, "clip_ratio/region_mean": 0.00849667435977608, "reward_total_mean": 0.9993604421615601, "reward_meter_mean": 0.9993604421615601, "reward_meter_std": 9.307587606599554e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993604421615601, "reward_total_composite_std": 9.307587606599554e-05} {"timestamp_utc": "2026-04-12T02:00:31Z", "mode": "eval", "global_step": 2400, "epoch": 0.09639715628388963, "eval_loss": NaN, "eval_runtime": 71.1062, "eval_samples_per_second": 1.463, "eval_steps_per_second": 0.183, "eval_num_tokens": 5428869.0, "eval_completions/mean_length": 196.8653846153846, "eval_completions/min_length": 59.46153846153846, "eval_completions/max_length": 372.46153846153845, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 196.8653846153846, "eval_completions/min_terminated_length": 59.46153846153846, "eval_completions/max_terminated_length": 372.46153846153845, "eval_rewards/meter/mean": 0.7441063798390902, "eval_rewards/meter/std": 0.3844332993030548, "eval_rewards/count_adherence/mean": 0.9244015262677119, "eval_rewards/count_adherence/std": 0.09978143326365031, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/repeat_penalty/mean": 0.8776137416179364, "eval_rewards/repeat_penalty/std": 0.1198913437815813, "eval_rewards/total_composite/mean": 0.6122724092923678, "eval_rewards/total_composite/std": 0.35310536279128146, "eval_reward": 0.6122724092923678, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.01914287731051445, "eval_sampling/sampling_logp_difference/max": 1.0589995751014123, "eval_sampling/importance_sampling_ratio/min": 0.36076818865079147, "eval_sampling/importance_sampling_ratio/mean": 1.004656663307777, "eval_sampling/importance_sampling_ratio/max": 1.4638345608344445, "eval_entropy": 0.19041176713429964, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6122724092923678, "eval_reward_meter_mean": 0.7441063798390902, "eval_reward_meter_std": 0.3844332993030548, "eval_reward_count_adherence_mean": 0.9244015262677119, "eval_reward_count_adherence_std": 0.09978143326365031, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_repeat_penalty_mean": 0.8776137416179364, "eval_reward_repeat_penalty_std": 0.1198913437815813, "eval_reward_total_composite_mean": 0.6122724092923678, "eval_reward_total_composite_std": 0.35310536279128146} {"timestamp_utc": "2026-04-12T02:00:39Z", "mode": "train", "global_step": 2401, "epoch": 0.09643732176567459, "loss": 0.0072, "grad_norm": 2.6642069816589355, "learning_rate": 2.7272727272727272e-06, "num_tokens": 5430680.0, "completions/mean_length": 65.375, "completions/min_length": 65.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.375, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9385327100753784, "rewards/meter/std": 0.10577936470508575, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9385327100753784, "rewards/total_composite/std": 0.10577936470508575, "reward": 0.9385327100753784, "reward_std": 0.10577936470508575, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02753767929971218, "sampling/sampling_logp_difference/max": 0.5753829479217529, "sampling/importance_sampling_ratio/min": 0.56248939037323, "sampling/importance_sampling_ratio/mean": 1.0091955661773682, "sampling/importance_sampling_ratio/max": 1.7399368286132812, "entropy": 0.1730100642889738, "clip_ratio/low_mean": 0.009360859869048, "clip_ratio/low_min": 0.009360859869048, "clip_ratio/high_mean": 0.01538461574818939, "clip_ratio/high_max": 0.01538461574818939, "clip_ratio/region_mean": 0.02474547561723739, "reward_total_mean": 0.9385327100753784, "reward_meter_mean": 0.9385327100753784, "reward_meter_std": 0.10577936470508575, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9385327100753784, "reward_total_composite_std": 0.10577936470508575} {"timestamp_utc": "2026-04-12T02:00:44Z", "mode": "train", "global_step": 2402, "epoch": 0.09647748724745954, "loss": 0.0097, "grad_norm": 5.85192346572876, "learning_rate": 2.724242424242424e-06, "num_tokens": 5432706.0, "completions/mean_length": 87.25, "completions/min_length": 84.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.25, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9501332640647888, "rewards/meter/std": 0.0800556018948555, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9501332640647888, "rewards/total_composite/std": 0.0800556018948555, "reward": 0.9501332640647888, "reward_std": 0.0800556018948555, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03878026455640793, "sampling/sampling_logp_difference/max": 1.6748466491699219, "sampling/importance_sampling_ratio/min": 0.18733689188957214, "sampling/importance_sampling_ratio/mean": 0.9965555667877197, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17476310580968857, "clip_ratio/low_mean": 0.007134926971048117, "clip_ratio/low_min": 0.007134926971048117, "clip_ratio/high_mean": 0.028704919386655092, "clip_ratio/high_max": 0.028704919386655092, "clip_ratio/region_mean": 0.03583984635770321, "reward_total_mean": 0.9501332640647888, "reward_meter_mean": 0.9501332640647888, "reward_meter_std": 0.0800556018948555, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9501332640647888, "reward_total_composite_std": 0.0800556018948555} {"timestamp_utc": "2026-04-12T02:00:49Z", "mode": "train", "global_step": 2403, "epoch": 0.0965176527292445, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.7212121212121213e-06, "num_tokens": 5434442.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00035840104101225734, "sampling/sampling_logp_difference/max": 0.004777892027050257, "sampling/importance_sampling_ratio/min": 0.9962913393974304, "sampling/importance_sampling_ratio/mean": 1.0003162622451782, "sampling/importance_sampling_ratio/max": 1.0047893524169922, "entropy": 0.0028671720647253096, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:00:59Z", "mode": "train", "global_step": 2404, "epoch": 0.09655781821102945, "loss": -0.1192, "grad_norm": 0.281558632850647, "learning_rate": 2.718181818181818e-06, "num_tokens": 5437697.0, "completions/mean_length": 506.875, "completions/min_length": 489.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.625, "completions/mean_terminated_length": 498.3333435058594, "completions/min_terminated_length": 489.0, "completions/max_terminated_length": 512.0, "rewards/meter/mean": 0.9645444750785828, "rewards/meter/std": 0.09066132456064224, "rewards/count_adherence/mean": 0.6907894611358643, "rewards/count_adherence/std": 0.01860806532204151, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9167022705078125, "rewards/repeat_penalty/std": 0.039733175188302994, "rewards/total_composite/mean": 0.6099640130996704, "rewards/total_composite/std": 0.05774744227528572, "reward": 0.6099640130996704, "reward_std": 0.057747457176446915, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03504529967904091, "sampling/sampling_logp_difference/max": 2.621081829071045, "sampling/importance_sampling_ratio/min": 0.07272414118051529, "sampling/importance_sampling_ratio/mean": 1.0061771869659424, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08459514752030373, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.010291029699146748, "clip_ratio/high_max": 0.010291029699146748, "clip_ratio/region_mean": 0.010291029699146748, "reward_total_mean": 0.6099640130996704, "reward_meter_mean": 0.9645444750785828, "reward_meter_std": 0.09066132456064224, "reward_count_adherence_mean": 0.6907894611358643, "reward_count_adherence_std": 0.01860806532204151, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9167022705078125, "reward_repeat_penalty_std": 0.039733175188302994, "reward_total_composite_mean": 0.6099640130996704, "reward_total_composite_std": 0.05774744227528572} {"timestamp_utc": "2026-04-12T02:01:04Z", "mode": "train", "global_step": 2405, "epoch": 0.0965979836928144, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.715151515151516e-06, "num_tokens": 5439465.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00042045500595122576, "sampling/sampling_logp_difference/max": 0.008271539583802223, "sampling/importance_sampling_ratio/min": 0.9947773814201355, "sampling/importance_sampling_ratio/mean": 1.0003942251205444, "sampling/importance_sampling_ratio/max": 1.0083059072494507, "entropy": 0.003762025124160573, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:01:08Z", "mode": "train", "global_step": 2406, "epoch": 0.09663814917459936, "loss": 0.0059, "grad_norm": 4.301211357116699, "learning_rate": 2.7121212121212127e-06, "num_tokens": 5441252.0, "completions/mean_length": 67.375, "completions/min_length": 67.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.375, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9776328802108765, "rewards/meter/std": 0.05134515464305878, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9776328802108765, "rewards/total_composite/std": 0.05134515464305878, "reward": 0.9776328802108765, "reward_std": 0.051345136016607285, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02041078545153141, "sampling/sampling_logp_difference/max": 1.1301698684692383, "sampling/importance_sampling_ratio/min": 0.3229783773422241, "sampling/importance_sampling_ratio/mean": 1.0020570755004883, "sampling/importance_sampling_ratio/max": 1.4865719079971313, "entropy": 0.13846461195498705, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.005542142200283706, "clip_ratio/high_max": 0.005542142200283706, "clip_ratio/region_mean": 0.005542142200283706, "reward_total_mean": 0.9776328802108765, "reward_meter_mean": 0.9776328802108765, "reward_meter_std": 0.05134515464305878, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9776328802108765, "reward_total_composite_std": 0.05134515464305878} {"timestamp_utc": "2026-04-12T02:01:13Z", "mode": "train", "global_step": 2407, "epoch": 0.09667831465638431, "loss": 0.0436, "grad_norm": 13.151045799255371, "learning_rate": 2.7090909090909095e-06, "num_tokens": 5443171.0, "completions/mean_length": 65.875, "completions/min_length": 61.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.875, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9381490349769592, "rewards/meter/std": 0.009724577888846397, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9381490349769592, "rewards/total_composite/std": 0.009724577888846397, "reward": 0.9381490349769592, "reward_std": 0.009724575094878674, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08292688429355621, "sampling/sampling_logp_difference/max": 2.107654094696045, "sampling/importance_sampling_ratio/min": 0.12152271717786789, "sampling/importance_sampling_ratio/mean": 0.9993854761123657, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.38669581711292267, "clip_ratio/low_mean": 0.020202020648866892, "clip_ratio/low_min": 0.020202020648866892, "clip_ratio/high_mean": 0.06410424876958132, "clip_ratio/high_max": 0.06410424876958132, "clip_ratio/region_mean": 0.08430626941844821, "reward_total_mean": 0.9381490349769592, "reward_meter_mean": 0.9381490349769592, "reward_meter_std": 0.009724577888846397, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9381490349769592, "reward_total_composite_std": 0.009724577888846397} {"timestamp_utc": "2026-04-12T02:01:18Z", "mode": "train", "global_step": 2408, "epoch": 0.09671848013816926, "loss": -0.0002, "grad_norm": 0.9304388165473938, "learning_rate": 2.7060606060606063e-06, "num_tokens": 5445313.0, "completions/mean_length": 97.75, "completions/min_length": 97.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.75, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9989974498748779, "rewards/meter/std": 0.0009041029843501747, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989974498748779, "rewards/total_composite/std": 0.0009041029843501747, "reward": 0.9989974498748779, "reward_std": 0.0009041179437190294, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012694788165390491, "sampling/sampling_logp_difference/max": 1.2907949686050415, "sampling/importance_sampling_ratio/min": 0.2750520408153534, "sampling/importance_sampling_ratio/mean": 0.9973183274269104, "sampling/importance_sampling_ratio/max": 1.6353120803833008, "entropy": 0.04637799598276615, "clip_ratio/low_mean": 0.002577319508418441, "clip_ratio/low_min": 0.002577319508418441, "clip_ratio/high_mean": 0.01663423073478043, "clip_ratio/high_max": 0.01663423073478043, "clip_ratio/region_mean": 0.01921155024319887, "reward_total_mean": 0.9989974498748779, "reward_meter_mean": 0.9989974498748779, "reward_meter_std": 0.0009041029843501747, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9989974498748779, "reward_total_composite_std": 0.0009041029843501747} {"timestamp_utc": "2026-04-12T02:01:24Z", "mode": "train", "global_step": 2409, "epoch": 0.09675864561995422, "loss": 0.0069, "grad_norm": 2.528715133666992, "learning_rate": 2.7030303030303036e-06, "num_tokens": 5448370.0, "completions/mean_length": 154.125, "completions/min_length": 150.0, "completions/max_length": 160.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 154.125, "completions/min_terminated_length": 150.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.997596800327301, "rewards/meter/std": 0.0010310488287359476, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.9619561433792114, "rewards/total_composite/std": 0.06579122692346573, "reward": 0.9619561433792114, "reward_std": 0.06579122692346573, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028948156163096428, "sampling/sampling_logp_difference/max": 1.3726117610931396, "sampling/importance_sampling_ratio/min": 0.2534441351890564, "sampling/importance_sampling_ratio/mean": 1.004657506942749, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.21369013004004955, "clip_ratio/low_mean": 0.001623376621864736, "clip_ratio/low_min": 0.001623376621864736, "clip_ratio/high_mean": 0.02023474802263081, "clip_ratio/high_max": 0.02023474802263081, "clip_ratio/region_mean": 0.021858124644495547, "reward_total_mean": 0.9619561433792114, "reward_meter_mean": 0.997596800327301, "reward_meter_std": 0.0010310488287359476, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.9619561433792114, "reward_total_composite_std": 0.06579122692346573} {"timestamp_utc": "2026-04-12T02:01:29Z", "mode": "train", "global_step": 2410, "epoch": 0.09679881110173917, "loss": -0.0112, "grad_norm": 2.699066400527954, "learning_rate": 2.7000000000000004e-06, "num_tokens": 5450115.0, "completions/mean_length": 67.125, "completions/min_length": 65.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.125, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9942950010299683, "rewards/meter/std": 0.0035084497649222612, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9942950010299683, "rewards/total_composite/std": 0.0035084497649222612, "reward": 0.9942950010299683, "reward_std": 0.003508440451696515, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02465406060218811, "sampling/sampling_logp_difference/max": 1.0274019241333008, "sampling/importance_sampling_ratio/min": 0.35793569684028625, "sampling/importance_sampling_ratio/mean": 1.0015225410461426, "sampling/importance_sampling_ratio/max": 1.9520822763442993, "entropy": 0.1753720510751009, "clip_ratio/low_mean": 0.007634902489371598, "clip_ratio/low_min": 0.007634902489371598, "clip_ratio/high_mean": 0.016363060451112688, "clip_ratio/high_max": 0.016363060451112688, "clip_ratio/region_mean": 0.023997962940484285, "reward_total_mean": 0.9942950010299683, "reward_meter_mean": 0.9942950010299683, "reward_meter_std": 0.0035084497649222612, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9942950010299683, "reward_total_composite_std": 0.0035084497649222612} {"timestamp_utc": "2026-04-12T02:01:34Z", "mode": "train", "global_step": 2411, "epoch": 0.09683897658352413, "loss": 0.0472, "grad_norm": 7.062148571014404, "learning_rate": 2.6969696969696972e-06, "num_tokens": 5452646.0, "completions/mean_length": 135.375, "completions/min_length": 130.0, "completions/max_length": 152.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 135.375, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 152.0, "rewards/meter/mean": 0.9934861660003662, "rewards/meter/std": 0.012767082080245018, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9934861660003662, "rewards/total_composite/std": 0.012767082080245018, "reward": 0.9934861660003662, "reward_std": 0.012767070904374123, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05575111508369446, "sampling/sampling_logp_difference/max": 1.5884981155395508, "sampling/importance_sampling_ratio/min": 0.2042321115732193, "sampling/importance_sampling_ratio/mean": 1.0023579597473145, "sampling/importance_sampling_ratio/max": 1.7682102918624878, "entropy": 0.38587315008044243, "clip_ratio/low_mean": 0.007401315961033106, "clip_ratio/low_min": 0.007401315961033106, "clip_ratio/high_mean": 0.043241268722340465, "clip_ratio/high_max": 0.043241268722340465, "clip_ratio/region_mean": 0.05064258468337357, "reward_total_mean": 0.9934861660003662, "reward_meter_mean": 0.9934861660003662, "reward_meter_std": 0.012767082080245018, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9934861660003662, "reward_total_composite_std": 0.012767082080245018} {"timestamp_utc": "2026-04-12T02:01:40Z", "mode": "train", "global_step": 2412, "epoch": 0.09687914206530908, "loss": 0.0204, "grad_norm": 3.8710875511169434, "learning_rate": 2.6939393939393945e-06, "num_tokens": 5454382.0, "completions/mean_length": 58.0, "completions/min_length": 55.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9953839778900146, "rewards/meter/std": 0.0009903458412736654, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9953839778900146, "rewards/total_composite/std": 0.0009903458412736654, "reward": 0.9953839778900146, "reward_std": 0.0009903304744511843, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02391960285604, "sampling/sampling_logp_difference/max": 1.1877875328063965, "sampling/importance_sampling_ratio/min": 0.44029319286346436, "sampling/importance_sampling_ratio/mean": 1.004859447479248, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11028503353009, "clip_ratio/low_mean": 0.015032077208161354, "clip_ratio/low_min": 0.015032077208161354, "clip_ratio/high_mean": 0.017077806871384382, "clip_ratio/high_max": 0.017077806871384382, "clip_ratio/region_mean": 0.032109884079545736, "reward_total_mean": 0.9953839778900146, "reward_meter_mean": 0.9953839778900146, "reward_meter_std": 0.0009903458412736654, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9953839778900146, "reward_total_composite_std": 0.0009903458412736654} {"timestamp_utc": "2026-04-12T02:01:48Z", "mode": "train", "global_step": 2413, "epoch": 0.09691930754709403, "loss": -0.0163, "grad_norm": 1.422156810760498, "learning_rate": 2.6909090909090913e-06, "num_tokens": 5459457.0, "completions/mean_length": 345.375, "completions/min_length": 318.0, "completions/max_length": 355.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 345.375, "completions/min_terminated_length": 318.0, "completions/max_terminated_length": 355.0, "rewards/meter/mean": 0.9982728958129883, "rewards/meter/std": 0.0010707140900194645, "rewards/count_adherence/mean": 0.887499988079071, "rewards/count_adherence/std": 0.035355325788259506, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9245098233222961, "rewards/repeat_penalty/std": 0.06273634731769562, "rewards/total_composite/mean": 0.8197740316390991, "rewards/total_composite/std": 0.07304169237613678, "reward": 0.8197740316390991, "reward_std": 0.07304168492555618, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031854599714279175, "sampling/sampling_logp_difference/max": 0.9705367088317871, "sampling/importance_sampling_ratio/min": 0.3788796067237854, "sampling/importance_sampling_ratio/mean": 1.0035998821258545, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2573400232940912, "clip_ratio/low_mean": 0.006614301120862365, "clip_ratio/low_min": 0.006614301120862365, "clip_ratio/high_mean": 0.02032158919610083, "clip_ratio/high_max": 0.02032158919610083, "clip_ratio/region_mean": 0.026935890316963196, "reward_total_mean": 0.8197740316390991, "reward_meter_mean": 0.9982728958129883, "reward_meter_std": 0.0010707140900194645, "reward_count_adherence_mean": 0.887499988079071, "reward_count_adherence_std": 0.035355325788259506, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9245098233222961, "reward_repeat_penalty_std": 0.06273634731769562, "reward_total_composite_mean": 0.8197740316390991, "reward_total_composite_std": 0.07304169237613678} {"timestamp_utc": "2026-04-12T02:01:53Z", "mode": "train", "global_step": 2414, "epoch": 0.09695947302887899, "loss": -0.0, "grad_norm": 0.009874632582068443, "learning_rate": 2.687878787878788e-06, "num_tokens": 5461241.0, "completions/mean_length": 67.0, "completions/min_length": 67.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981490969657898, "rewards/meter/std": 8.935132427723147e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981490969657898, "rewards/total_composite/std": 8.935132427723147e-06, "reward": 0.9981490969657898, "reward_std": 8.935131518228445e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0008906561997719109, "sampling/sampling_logp_difference/max": 0.16530299186706543, "sampling/importance_sampling_ratio/min": 0.980025053024292, "sampling/importance_sampling_ratio/mean": 1.0007779598236084, "sampling/importance_sampling_ratio/max": 1.1797505617141724, "entropy": 0.006271830061450601, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0018656715983524919, "reward_total_mean": 0.9981490969657898, "reward_meter_mean": 0.9981490969657898, "reward_meter_std": 8.935132427723147e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981490969657898, "reward_total_composite_std": 8.935132427723147e-06} {"timestamp_utc": "2026-04-12T02:01:58Z", "mode": "train", "global_step": 2415, "epoch": 0.09699963851066394, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.684848484848485e-06, "num_tokens": 5462745.0, "completions/mean_length": 35.0, "completions/min_length": 35.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9980231523513794, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980231523513794, "rewards/total_composite/std": 0.0, "reward": 0.9980231523513794, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00018749816808849573, "sampling/sampling_logp_difference/max": 0.003093225881457329, "sampling/importance_sampling_ratio/min": 0.9994993209838867, "sampling/importance_sampling_ratio/mean": 1.000183343887329, "sampling/importance_sampling_ratio/max": 1.0030980110168457, "entropy": 0.001466380010242574, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9980231523513794, "reward_meter_mean": 0.9980231523513794, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9980231523513794, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:02:03Z", "mode": "train", "global_step": 2416, "epoch": 0.0970398039924489, "loss": 0.0002, "grad_norm": 0.24654512107372284, "learning_rate": 2.6818181818181822e-06, "num_tokens": 5464841.0, "completions/mean_length": 98.0, "completions/min_length": 98.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.0, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9980032444000244, "rewards/meter/std": 8.327968316734768e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980032444000244, "rewards/total_composite/std": 8.327968316734768e-06, "reward": 0.9980032444000244, "reward_std": 8.322024768858682e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0037318698596209288, "sampling/sampling_logp_difference/max": 0.787595272064209, "sampling/importance_sampling_ratio/min": 0.45493748784065247, "sampling/importance_sampling_ratio/mean": 1.0002275705337524, "sampling/importance_sampling_ratio/max": 1.1473532915115356, "entropy": 0.023096754681319, "clip_ratio/low_mean": 0.0012755101779475808, "clip_ratio/low_min": 0.0012755101779475808, "clip_ratio/high_mean": 0.0012755101779475808, "clip_ratio/high_max": 0.0012755101779475808, "clip_ratio/region_mean": 0.0025510203558951616, "reward_total_mean": 0.9980032444000244, "reward_meter_mean": 0.9980032444000244, "reward_meter_std": 8.327968316734768e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9980032444000244, "reward_total_composite_std": 8.327968316734768e-06} {"timestamp_utc": "2026-04-12T02:02:07Z", "mode": "train", "global_step": 2417, "epoch": 0.09707996947423385, "loss": 0.0088, "grad_norm": 3.3394386768341064, "learning_rate": 2.678787878787879e-06, "num_tokens": 5466420.0, "completions/mean_length": 35.375, "completions/min_length": 34.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.375, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.999369204044342, "rewards/meter/std": 0.0002697483287192881, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999369204044342, "rewards/total_composite/std": 0.0002697483287192881, "reward": 0.999369204044342, "reward_std": 0.00026974434149451554, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02549952082335949, "sampling/sampling_logp_difference/max": 0.7178728580474854, "sampling/importance_sampling_ratio/min": 0.48778876662254333, "sampling/importance_sampling_ratio/mean": 1.0062538385391235, "sampling/importance_sampling_ratio/max": 1.802677035331726, "entropy": 0.18990459106862545, "clip_ratio/low_mean": 0.006756756920367479, "clip_ratio/low_min": 0.006756756920367479, "clip_ratio/high_mean": 0.02460317499935627, "clip_ratio/high_max": 0.02460317499935627, "clip_ratio/region_mean": 0.03135993191972375, "reward_total_mean": 0.999369204044342, "reward_meter_mean": 0.999369204044342, "reward_meter_std": 0.0002697483287192881, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.999369204044342, "reward_total_composite_std": 0.0002697483287192881} {"timestamp_utc": "2026-04-12T02:02:12Z", "mode": "train", "global_step": 2418, "epoch": 0.0971201349560188, "loss": 0.0167, "grad_norm": 11.102910041809082, "learning_rate": 2.675757575757576e-06, "num_tokens": 5468444.0, "completions/mean_length": 82.0, "completions/min_length": 79.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.0, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.8613059520721436, "rewards/meter/std": 0.17161336541175842, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9285714626312256, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.8016860485076904, "rewards/total_composite/std": 0.17636200785636902, "reward": 0.8016860485076904, "reward_std": 0.1763620227575302, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08092934638261795, "sampling/sampling_logp_difference/max": 2.628443717956543, "sampling/importance_sampling_ratio/min": 0.07219072431325912, "sampling/importance_sampling_ratio/mean": 0.9934130311012268, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3536304421722889, "clip_ratio/low_mean": 0.012442129664123058, "clip_ratio/low_min": 0.012442129664123058, "clip_ratio/high_mean": 0.05160611355677247, "clip_ratio/high_max": 0.05160611355677247, "clip_ratio/region_mean": 0.06404824322089553, "reward_total_mean": 0.8016860485076904, "reward_meter_mean": 0.8613059520721436, "reward_meter_std": 0.17161336541175842, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9285714626312256, "reward_repeat_penalty_std": 0.07636035233736038, "reward_total_composite_mean": 0.8016860485076904, "reward_total_composite_std": 0.17636200785636902} {"timestamp_utc": "2026-04-12T02:02:17Z", "mode": "train", "global_step": 2419, "epoch": 0.09716030043780376, "loss": -0.0001, "grad_norm": 0.25810882449150085, "learning_rate": 2.6727272727272727e-06, "num_tokens": 5470156.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.7876332998275757, "rewards/meter/std": 1.5573279597447254e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7876332998275757, "rewards/total_composite/std": 1.5573279597447254e-05, "reward": 0.7876332998275757, "reward_std": 1.558229632792063e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.002996515017002821, "sampling/sampling_logp_difference/max": 1.1518197059631348, "sampling/importance_sampling_ratio/min": 0.3160611093044281, "sampling/importance_sampling_ratio/mean": 0.9987244606018066, "sampling/importance_sampling_ratio/max": 1.006128191947937, "entropy": 0.00269945221953094, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7876332998275757, "reward_meter_mean": 0.7876332998275757, "reward_meter_std": 1.5573279597447254e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7876332998275757, "reward_total_composite_std": 1.5573279597447254e-05} {"timestamp_utc": "2026-04-12T02:02:23Z", "mode": "train", "global_step": 2420, "epoch": 0.09720046591958871, "loss": 0.005, "grad_norm": 3.2308475971221924, "learning_rate": 2.66969696969697e-06, "num_tokens": 5472925.0, "completions/mean_length": 169.125, "completions/min_length": 163.0, "completions/max_length": 176.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 169.125, "completions/min_terminated_length": 163.0, "completions/max_terminated_length": 176.0, "rewards/meter/mean": 0.9988895654678345, "rewards/meter/std": 0.00021211641433183104, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333134651184, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.9572727680206299, "rewards/total_composite/std": 0.057507775723934174, "reward": 0.9572727680206299, "reward_std": 0.05750780552625656, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04539603367447853, "sampling/sampling_logp_difference/max": 1.4913952350616455, "sampling/importance_sampling_ratio/min": 0.2250584363937378, "sampling/importance_sampling_ratio/mean": 1.0061159133911133, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3278270773589611, "clip_ratio/low_mean": 0.011130106868222356, "clip_ratio/low_min": 0.011130106868222356, "clip_ratio/high_mean": 0.028763922629877925, "clip_ratio/high_max": 0.028763922629877925, "clip_ratio/region_mean": 0.03989402949810028, "reward_total_mean": 0.9572727680206299, "reward_meter_mean": 0.9988895654678345, "reward_meter_std": 0.00021211641433183104, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333134651184, "reward_repeat_penalty_std": 0.05750546231865883, "reward_total_composite_mean": 0.9572727680206299, "reward_total_composite_std": 0.057507775723934174} {"timestamp_utc": "2026-04-12T02:02:28Z", "mode": "train", "global_step": 2421, "epoch": 0.09724063140137366, "loss": -0.0028, "grad_norm": 0.6984437704086304, "learning_rate": 2.666666666666667e-06, "num_tokens": 5475014.0, "completions/mean_length": 106.125, "completions/min_length": 105.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.125, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.9992530345916748, "rewards/meter/std": 0.00013918116746935993, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992530345916748, "rewards/total_composite/std": 0.00013918116746935993, "reward": 0.9992530345916748, "reward_std": 0.0001391761179547757, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008111930452287197, "sampling/sampling_logp_difference/max": 0.6127934455871582, "sampling/importance_sampling_ratio/min": 0.5418351888656616, "sampling/importance_sampling_ratio/mean": 1.000658392906189, "sampling/importance_sampling_ratio/max": 1.1507465839385986, "entropy": 0.056452156975865364, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.004716981202363968, "clip_ratio/high_max": 0.004716981202363968, "clip_ratio/region_mean": 0.004716981202363968, "reward_total_mean": 0.9992530345916748, "reward_meter_mean": 0.9992530345916748, "reward_meter_std": 0.00013918116746935993, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992530345916748, "reward_total_composite_std": 0.00013918116746935993} {"timestamp_utc": "2026-04-12T02:02:34Z", "mode": "train", "global_step": 2422, "epoch": 0.09728079688315862, "loss": 0.0142, "grad_norm": 2.5043861865997314, "learning_rate": 2.6636363636363637e-06, "num_tokens": 5477412.0, "completions/mean_length": 134.75, "completions/min_length": 132.0, "completions/max_length": 137.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 134.75, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.9812095761299133, "rewards/meter/std": 0.03474520519375801, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.1079898476600647, "rewards/total_composite/mean": 0.841211199760437, "rewards/total_composite/std": 0.1127474382519722, "reward": 0.841211199760437, "reward_std": 0.1127474457025528, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026558201760053635, "sampling/sampling_logp_difference/max": 1.1920127868652344, "sampling/importance_sampling_ratio/min": 0.36081838607788086, "sampling/importance_sampling_ratio/mean": 1.0124142169952393, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23221887275576591, "clip_ratio/low_mean": 0.010076848906464875, "clip_ratio/low_min": 0.010076848906464875, "clip_ratio/high_mean": 0.012170514499302953, "clip_ratio/high_max": 0.012170514499302953, "clip_ratio/region_mean": 0.022247363405767828, "reward_total_mean": 0.841211199760437, "reward_meter_mean": 0.9812095761299133, "reward_meter_std": 0.03474520519375801, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.1079898476600647, "reward_total_composite_mean": 0.841211199760437, "reward_total_composite_std": 0.1127474382519722} {"timestamp_utc": "2026-04-12T02:02:38Z", "mode": "train", "global_step": 2423, "epoch": 0.09732096236494357, "loss": 0.0077, "grad_norm": 5.193559169769287, "learning_rate": 2.660606060606061e-06, "num_tokens": 5479245.0, "completions/mean_length": 64.125, "completions/min_length": 64.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.125, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9993478059768677, "rewards/meter/std": 0.0001315748959314078, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993478059768677, "rewards/total_composite/std": 0.0001315748959314078, "reward": 0.9993478059768677, "reward_std": 0.00013158157526049763, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006412571296095848, "sampling/sampling_logp_difference/max": 1.5588197708129883, "sampling/importance_sampling_ratio/min": 0.21038421988487244, "sampling/importance_sampling_ratio/mean": 0.9980660080909729, "sampling/importance_sampling_ratio/max": 1.234817385673523, "entropy": 0.021533598424866796, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.009765625, "clip_ratio/high_max": 0.009765625, "clip_ratio/region_mean": 0.009765625, "reward_total_mean": 0.9993478059768677, "reward_meter_mean": 0.9993478059768677, "reward_meter_std": 0.0001315748959314078, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993478059768677, "reward_total_composite_std": 0.0001315748959314078} {"timestamp_utc": "2026-04-12T02:02:48Z", "mode": "train", "global_step": 2424, "epoch": 0.09736112784672853, "loss": 0.1719, "grad_norm": 1.59493887424469, "learning_rate": 2.6575757575757577e-06, "num_tokens": 5481641.0, "completions/mean_length": 187.5, "completions/min_length": 77.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 79.33333587646484, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.9973663687705994, "rewards/meter/std": 0.0009385403245687485, "rewards/count_adherence/mean": 0.23749999701976776, "rewards/count_adherence/std": 0.25460052490234375, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9709615707397461, "rewards/repeat_penalty/std": 0.06743928045034409, "rewards/total_composite/mean": 0.21816782653331757, "rewards/total_composite/std": 0.22094322741031647, "reward": 0.21816782653331757, "reward_std": 0.22094321250915527, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.041650205850601196, "sampling/sampling_logp_difference/max": 1.7469613552093506, "sampling/importance_sampling_ratio/min": 0.17430277168750763, "sampling/importance_sampling_ratio/mean": 0.99797523021698, "sampling/importance_sampling_ratio/max": 1.734250783920288, "entropy": 0.1750364489853382, "clip_ratio/low_mean": 0.022042680298909545, "clip_ratio/low_min": 0.022042680298909545, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.022042680298909545, "reward_total_mean": 0.21816782653331757, "reward_meter_mean": 0.9973663687705994, "reward_meter_std": 0.0009385403245687485, "reward_count_adherence_mean": 0.23749999701976776, "reward_count_adherence_std": 0.25460052490234375, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9709615707397461, "reward_repeat_penalty_std": 0.06743928045034409, "reward_total_composite_mean": 0.21816782653331757, "reward_total_composite_std": 0.22094322741031647} {"timestamp_utc": "2026-04-12T02:02:53Z", "mode": "train", "global_step": 2425, "epoch": 0.09740129332851348, "loss": 0.0094, "grad_norm": 3.781071186065674, "learning_rate": 2.6545454545454546e-06, "num_tokens": 5483726.0, "completions/mean_length": 101.625, "completions/min_length": 98.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.625, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.8708481192588806, "rewards/meter/std": 0.34927064180374146, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8708481192588806, "rewards/total_composite/std": 0.34927064180374146, "reward": 0.8708481192588806, "reward_std": 0.34927064180374146, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02785804308950901, "sampling/sampling_logp_difference/max": 0.8949475288391113, "sampling/importance_sampling_ratio/min": 0.40862902998924255, "sampling/importance_sampling_ratio/mean": 1.0064235925674438, "sampling/importance_sampling_ratio/max": 1.9378726482391357, "entropy": 0.20133807510137558, "clip_ratio/low_mean": 0.0036764706019312143, "clip_ratio/low_min": 0.0036764706019312143, "clip_ratio/high_mean": 0.017109742388129234, "clip_ratio/high_max": 0.017109742388129234, "clip_ratio/region_mean": 0.02078621299006045, "reward_total_mean": 0.8708481192588806, "reward_meter_mean": 0.8708481192588806, "reward_meter_std": 0.34927064180374146, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8708481192588806, "reward_total_composite_std": 0.34927064180374146} {"timestamp_utc": "2026-04-12T02:02:58Z", "mode": "train", "global_step": 2426, "epoch": 0.09744145881029843, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.6515151515151514e-06, "num_tokens": 5485422.0, "completions/mean_length": 67.0, "completions/min_length": 67.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981522560119629, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981522560119629, "rewards/total_composite/std": 0.0, "reward": 0.9981522560119629, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00044331722892820835, "sampling/sampling_logp_difference/max": 0.03401622548699379, "sampling/importance_sampling_ratio/min": 0.9665557742118835, "sampling/importance_sampling_ratio/mean": 1.0000972747802734, "sampling/importance_sampling_ratio/max": 1.0289011001586914, "entropy": 0.004820480738999322, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9981522560119629, "reward_meter_mean": 0.9981522560119629, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981522560119629, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:03:02Z", "mode": "train", "global_step": 2427, "epoch": 0.09748162429208339, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.6484848484848487e-06, "num_tokens": 5487246.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0009743176633492112, "sampling/sampling_logp_difference/max": 0.0836307555437088, "sampling/importance_sampling_ratio/min": 0.9197707772254944, "sampling/importance_sampling_ratio/mean": 1.0001752376556396, "sampling/importance_sampling_ratio/max": 1.0430419445037842, "entropy": 0.011770866345614195, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:03:06Z", "mode": "train", "global_step": 2428, "epoch": 0.09752178977386834, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.6454545454545455e-06, "num_tokens": 5488686.0, "completions/mean_length": 28.0, "completions/min_length": 28.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.0, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9943599104881287, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9943599104881287, "rewards/total_composite/std": 0.0, "reward": 0.9943599104881287, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.007659213151782751, "sampling/sampling_logp_difference/max": 0.8227400779724121, "sampling/importance_sampling_ratio/min": 0.4392264783382416, "sampling/importance_sampling_ratio/mean": 0.9964240789413452, "sampling/importance_sampling_ratio/max": 1.0674165487289429, "entropy": 0.01067307055927813, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9943599104881287, "reward_meter_mean": 0.9943599104881287, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9943599104881287, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:03:11Z", "mode": "train", "global_step": 2429, "epoch": 0.0975619552556533, "loss": 0.0, "grad_norm": 0.09379827976226807, "learning_rate": 2.6424242424242423e-06, "num_tokens": 5490494.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993983507156372, "rewards/meter/std": 3.034573182958411e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993983507156372, "rewards/total_composite/std": 3.034573182958411e-06, "reward": 0.9993983507156372, "reward_std": 3.034573182958411e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0012968340888619423, "sampling/sampling_logp_difference/max": 0.16098451614379883, "sampling/importance_sampling_ratio/min": 0.9115949869155884, "sampling/importance_sampling_ratio/mean": 1.000643253326416, "sampling/importance_sampling_ratio/max": 1.1746667623519897, "entropy": 0.011805565096437931, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993983507156372, "reward_meter_mean": 0.9993983507156372, "reward_meter_std": 3.034573182958411e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993983507156372, "reward_total_composite_std": 3.034573182958411e-06} {"timestamp_utc": "2026-04-12T02:03:16Z", "mode": "train", "global_step": 2430, "epoch": 0.09760212073743825, "loss": 0.0007, "grad_norm": 5.21270751953125, "learning_rate": 2.6393939393939396e-06, "num_tokens": 5492246.0, "completions/mean_length": 67.0, "completions/min_length": 65.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9988259077072144, "rewards/meter/std": 0.0006359845283441246, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9988259077072144, "rewards/total_composite/std": 0.0006359845283441246, "reward": 0.9988259077072144, "reward_std": 0.0006359879043884575, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05176051706075668, "sampling/sampling_logp_difference/max": 1.786757230758667, "sampling/importance_sampling_ratio/min": 0.16750246286392212, "sampling/importance_sampling_ratio/mean": 1.002232551574707, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2827076967805624, "clip_ratio/low_mean": 0.009443168761208653, "clip_ratio/low_min": 0.009443168761208653, "clip_ratio/high_mean": 0.02971602266188711, "clip_ratio/high_max": 0.02971602266188711, "clip_ratio/region_mean": 0.03915919142309576, "reward_total_mean": 0.9988259077072144, "reward_meter_mean": 0.9988259077072144, "reward_meter_std": 0.0006359845283441246, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9988259077072144, "reward_total_composite_std": 0.0006359845283441246} {"timestamp_utc": "2026-04-12T02:03:21Z", "mode": "train", "global_step": 2431, "epoch": 0.0976422862192232, "loss": 0.0117, "grad_norm": 3.1840193271636963, "learning_rate": 2.6363636363636364e-06, "num_tokens": 5494352.0, "completions/mean_length": 102.25, "completions/min_length": 98.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.25, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.998919665813446, "rewards/meter/std": 0.00028825365006923676, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998919665813446, "rewards/total_composite/std": 0.00028825365006923676, "reward": 0.998919665813446, "reward_std": 0.0002882632543332875, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03998712822794914, "sampling/sampling_logp_difference/max": 1.0248355865478516, "sampling/importance_sampling_ratio/min": 0.3685198724269867, "sampling/importance_sampling_ratio/mean": 1.0071525573730469, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3069701548665762, "clip_ratio/low_mean": 0.019234189530834556, "clip_ratio/low_min": 0.019234189530834556, "clip_ratio/high_mean": 0.023538942215964198, "clip_ratio/high_max": 0.023538942215964198, "clip_ratio/region_mean": 0.042773131746798754, "reward_total_mean": 0.998919665813446, "reward_meter_mean": 0.998919665813446, "reward_meter_std": 0.00028825365006923676, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.998919665813446, "reward_total_composite_std": 0.00028825365006923676} {"timestamp_utc": "2026-04-12T02:03:28Z", "mode": "train", "global_step": 2432, "epoch": 0.09768245170100816, "loss": 0.0053, "grad_norm": 0.9189408421516418, "learning_rate": 2.6333333333333332e-06, "num_tokens": 5497991.0, "completions/mean_length": 246.875, "completions/min_length": 245.0, "completions/max_length": 250.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 246.875, "completions/min_terminated_length": 245.0, "completions/max_terminated_length": 250.0, "rewards/meter/mean": 0.999138593673706, "rewards/meter/std": 0.00011935058864764869, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7307692766189575, "rewards/repeat_penalty/std": 0.041117113083601, "rewards/total_composite/mean": 0.7301368713378906, "rewards/total_composite/std": 0.041023507714271545, "reward": 0.7301368713378906, "reward_std": 0.041023530066013336, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013350401073694229, "sampling/sampling_logp_difference/max": 1.0539226531982422, "sampling/importance_sampling_ratio/min": 0.3485677242279053, "sampling/importance_sampling_ratio/mean": 1.0021779537200928, "sampling/importance_sampling_ratio/max": 1.3559163808822632, "entropy": 0.08605087455362082, "clip_ratio/low_mean": 0.0015080645098350942, "clip_ratio/low_min": 0.0015080645098350942, "clip_ratio/high_mean": 0.0035590012557804585, "clip_ratio/high_max": 0.0035590012557804585, "clip_ratio/region_mean": 0.005067065765615553, "reward_total_mean": 0.7301368713378906, "reward_meter_mean": 0.999138593673706, "reward_meter_std": 0.00011935058864764869, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7307692766189575, "reward_repeat_penalty_std": 0.041117113083601, "reward_total_composite_mean": 0.7301368713378906, "reward_total_composite_std": 0.041023507714271545} {"timestamp_utc": "2026-04-12T02:03:34Z", "mode": "train", "global_step": 2433, "epoch": 0.09772261718279311, "loss": -0.0002, "grad_norm": 0.08106043934822083, "learning_rate": 2.63030303030303e-06, "num_tokens": 5500582.0, "completions/mean_length": 176.875, "completions/min_length": 176.0, "completions/max_length": 177.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 176.875, "completions/min_terminated_length": 176.0, "completions/max_terminated_length": 177.0, "rewards/meter/mean": 0.9992530345916748, "rewards/meter/std": 1.491289003752172e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7777777910232544, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7771967649459839, "rewards/total_composite/std": 1.1599345270951744e-05, "reward": 0.7771967649459839, "reward_std": 1.1595264368224889e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007231563795357943, "sampling/sampling_logp_difference/max": 0.7993268966674805, "sampling/importance_sampling_ratio/min": 0.4496315121650696, "sampling/importance_sampling_ratio/mean": 1.002611517906189, "sampling/importance_sampling_ratio/max": 1.2289493083953857, "entropy": 0.05232963711023331, "clip_ratio/low_mean": 0.0021186440717428923, "clip_ratio/low_min": 0.0021186440717428923, "clip_ratio/high_mean": 0.0014204545877873898, "clip_ratio/high_max": 0.0014204545877873898, "clip_ratio/region_mean": 0.003539098659530282, "reward_total_mean": 0.7771967649459839, "reward_meter_mean": 0.9992530345916748, "reward_meter_std": 1.491289003752172e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7777777910232544, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7771967649459839, "reward_total_composite_std": 1.1599345270951744e-05} {"timestamp_utc": "2026-04-12T02:03:39Z", "mode": "train", "global_step": 2434, "epoch": 0.09776278266457807, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.6272727272727278e-06, "num_tokens": 5502654.0, "completions/mean_length": 67.0, "completions/min_length": 67.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981522560119629, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981522560119629, "rewards/total_composite/std": 0.0, "reward": 0.9981522560119629, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0005770151037722826, "sampling/sampling_logp_difference/max": 0.009044456295669079, "sampling/importance_sampling_ratio/min": 0.9965153932571411, "sampling/importance_sampling_ratio/mean": 1.0005393028259277, "sampling/importance_sampling_ratio/max": 1.0090855360031128, "entropy": 0.005529700662009418, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9981522560119629, "reward_meter_mean": 0.9981522560119629, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981522560119629, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:03:44Z", "mode": "train", "global_step": 2435, "epoch": 0.09780294814636302, "loss": 0.003, "grad_norm": 5.022347927093506, "learning_rate": 2.6242424242424246e-06, "num_tokens": 5504823.0, "completions/mean_length": 90.125, "completions/min_length": 84.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.125, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.9843810200691223, "rewards/meter/std": 0.03011130914092064, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9594758749008179, "rewards/total_composite/std": 0.07208064943552017, "reward": 0.9594758749008179, "reward_std": 0.07208065688610077, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03877629339694977, "sampling/sampling_logp_difference/max": 1.5140687227249146, "sampling/importance_sampling_ratio/min": 0.22001299262046814, "sampling/importance_sampling_ratio/mean": 1.0028375387191772, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2525391634553671, "clip_ratio/low_mean": 0.00696005008649081, "clip_ratio/low_min": 0.00696005008649081, "clip_ratio/high_mean": 0.03053560631815344, "clip_ratio/high_max": 0.03053560631815344, "clip_ratio/region_mean": 0.03749565640464425, "reward_total_mean": 0.9594758749008179, "reward_meter_mean": 0.9843810200691223, "reward_meter_std": 0.03011130914092064, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9594758749008179, "reward_total_composite_std": 0.07208064943552017} {"timestamp_utc": "2026-04-12T02:03:49Z", "mode": "train", "global_step": 2436, "epoch": 0.09784311362814797, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.621212121212122e-06, "num_tokens": 5506655.0, "completions/mean_length": 67.0, "completions/min_length": 67.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981522560119629, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981522560119629, "rewards/total_composite/std": 0.0, "reward": 0.9981522560119629, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0004700902500189841, "sampling/sampling_logp_difference/max": 0.009367566555738449, "sampling/importance_sampling_ratio/min": 0.9982850551605225, "sampling/importance_sampling_ratio/mean": 1.0004562139511108, "sampling/importance_sampling_ratio/max": 1.0094116926193237, "entropy": 0.004693651368143037, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9981522560119629, "reward_meter_mean": 0.9981522560119629, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981522560119629, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:03:54Z", "mode": "train", "global_step": 2437, "epoch": 0.09788327910993293, "loss": 0.006, "grad_norm": 0.9422140717506409, "learning_rate": 2.6181818181818187e-06, "num_tokens": 5508881.0, "completions/mean_length": 100.25, "completions/min_length": 98.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.25, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.9990655183792114, "rewards/meter/std": 0.00015276693738996983, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990655183792114, "rewards/total_composite/std": 0.00015276693738996983, "reward": 0.9990655183792114, "reward_std": 0.00015277891361620277, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010102971456944942, "sampling/sampling_logp_difference/max": 1.0181710720062256, "sampling/importance_sampling_ratio/min": 0.36125504970550537, "sampling/importance_sampling_ratio/mean": 0.9993890523910522, "sampling/importance_sampling_ratio/max": 1.4387589693069458, "entropy": 0.04010937362909317, "clip_ratio/low_mean": 0.00376287626568228, "clip_ratio/low_min": 0.00376287626568228, "clip_ratio/high_mean": 0.003763133892789483, "clip_ratio/high_max": 0.003763133892789483, "clip_ratio/region_mean": 0.007526010158471763, "reward_total_mean": 0.9990655183792114, "reward_meter_mean": 0.9990655183792114, "reward_meter_std": 0.00015276693738996983, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990655183792114, "reward_total_composite_std": 0.00015276693738996983} {"timestamp_utc": "2026-04-12T02:03:58Z", "mode": "train", "global_step": 2438, "epoch": 0.09792344459171788, "loss": 0.0011, "grad_norm": 1.3743865489959717, "learning_rate": 2.6151515151515155e-06, "num_tokens": 5510611.0, "completions/mean_length": 64.25, "completions/min_length": 64.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.25, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9993854761123657, "rewards/meter/std": 2.579814099590294e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993854761123657, "rewards/total_composite/std": 2.579814099590294e-05, "reward": 0.9993854761123657, "reward_std": 2.5807335987337865e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0030143873300403357, "sampling/sampling_logp_difference/max": 0.5796234607696533, "sampling/importance_sampling_ratio/min": 0.5601092576980591, "sampling/importance_sampling_ratio/mean": 1.0008904933929443, "sampling/importance_sampling_ratio/max": 1.3340281248092651, "entropy": 0.010751945606898516, "clip_ratio/low_mean": 0.001923076924867928, "clip_ratio/low_min": 0.001923076924867928, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.001923076924867928, "reward_total_mean": 0.9993854761123657, "reward_meter_mean": 0.9993854761123657, "reward_meter_std": 2.579814099590294e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993854761123657, "reward_total_composite_std": 2.579814099590294e-05} {"timestamp_utc": "2026-04-12T02:04:04Z", "mode": "train", "global_step": 2439, "epoch": 0.09796361007350284, "loss": 0.0054, "grad_norm": 3.004152774810791, "learning_rate": 2.6121212121212123e-06, "num_tokens": 5512728.0, "completions/mean_length": 98.625, "completions/min_length": 94.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.625, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.9990460872650146, "rewards/meter/std": 0.00018491577066015452, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990460872650146, "rewards/total_composite/std": 0.00018491577066015452, "reward": 0.9990460872650146, "reward_std": 0.00018490130605641752, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.039266087114810944, "sampling/sampling_logp_difference/max": 1.522506594657898, "sampling/importance_sampling_ratio/min": 0.2181643694639206, "sampling/importance_sampling_ratio/mean": 1.0018714666366577, "sampling/importance_sampling_ratio/max": 1.664224624633789, "entropy": 0.282865421846509, "clip_ratio/low_mean": 0.01263557211495936, "clip_ratio/low_min": 0.01263557211495936, "clip_ratio/high_mean": 0.02017652615904808, "clip_ratio/high_max": 0.02017652615904808, "clip_ratio/region_mean": 0.03281209827400744, "reward_total_mean": 0.9990460872650146, "reward_meter_mean": 0.9990460872650146, "reward_meter_std": 0.00018491577066015452, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990460872650146, "reward_total_composite_std": 0.00018491577066015452} {"timestamp_utc": "2026-04-12T02:04:12Z", "mode": "train", "global_step": 2440, "epoch": 0.09800377555528779, "loss": -0.0037, "grad_norm": 1.5796713829040527, "learning_rate": 2.6090909090909096e-06, "num_tokens": 5517263.0, "completions/mean_length": 344.875, "completions/min_length": 329.0, "completions/max_length": 358.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 344.875, "completions/min_terminated_length": 329.0, "completions/max_terminated_length": 358.0, "rewards/meter/mean": 0.9985554218292236, "rewards/meter/std": 0.00046954461140558124, "rewards/count_adherence/mean": 0.8181818127632141, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9117647409439087, "rewards/repeat_penalty/std": 0.04446640983223915, "rewards/total_composite/mean": 0.7449136972427368, "rewards/total_composite/std": 0.036383431404829025, "reward": 0.7449136972427368, "reward_std": 0.036383435130119324, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030231278389692307, "sampling/sampling_logp_difference/max": 1.6795151233673096, "sampling/importance_sampling_ratio/min": 0.1864643543958664, "sampling/importance_sampling_ratio/mean": 1.0063951015472412, "sampling/importance_sampling_ratio/max": 1.8043537139892578, "entropy": 0.2484235055744648, "clip_ratio/low_mean": 0.00623403606005013, "clip_ratio/low_min": 0.00623403606005013, "clip_ratio/high_mean": 0.01647980441339314, "clip_ratio/high_max": 0.01647980441339314, "clip_ratio/region_mean": 0.02271384047344327, "reward_total_mean": 0.7449136972427368, "reward_meter_mean": 0.9985554218292236, "reward_meter_std": 0.00046954461140558124, "reward_count_adherence_mean": 0.8181818127632141, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9117647409439087, "reward_repeat_penalty_std": 0.04446640983223915, "reward_total_composite_mean": 0.7449136972427368, "reward_total_composite_std": 0.036383431404829025} {"timestamp_utc": "2026-04-12T02:04:17Z", "mode": "train", "global_step": 2441, "epoch": 0.09804394103707274, "loss": -0.0108, "grad_norm": 5.441035270690918, "learning_rate": 2.6060606060606064e-06, "num_tokens": 5519192.0, "completions/mean_length": 79.125, "completions/min_length": 76.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.125, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.9938977360725403, "rewards/meter/std": 0.009339890442788601, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8725272417068481, "rewards/total_composite/std": 0.35255616903305054, "reward": 0.8725272417068481, "reward_std": 0.35255616903305054, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03632466122508049, "sampling/sampling_logp_difference/max": 2.1114463806152344, "sampling/importance_sampling_ratio/min": 0.1210627406835556, "sampling/importance_sampling_ratio/mean": 0.9990738034248352, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.312020493671298, "clip_ratio/low_mean": 0.003289473708719015, "clip_ratio/low_min": 0.003289473708719015, "clip_ratio/high_mean": 0.025135348900221288, "clip_ratio/high_max": 0.025135348900221288, "clip_ratio/region_mean": 0.028424822608940303, "reward_total_mean": 0.8725272417068481, "reward_meter_mean": 0.9938977360725403, "reward_meter_std": 0.009339890442788601, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8725272417068481, "reward_total_composite_std": 0.35255616903305054} {"timestamp_utc": "2026-04-12T02:04:21Z", "mode": "train", "global_step": 2442, "epoch": 0.0980841065188577, "loss": -0.0071, "grad_norm": 1.337915062904358, "learning_rate": 2.6030303030303033e-06, "num_tokens": 5521213.0, "completions/mean_length": 71.625, "completions/min_length": 69.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.625, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9993535876274109, "rewards/meter/std": 0.00013377754657994956, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993535876274109, "rewards/total_composite/std": 0.00013377754657994956, "reward": 0.9993535876274109, "reward_std": 0.00013379388838075101, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013958643190562725, "sampling/sampling_logp_difference/max": 0.9804167747497559, "sampling/importance_sampling_ratio/min": 0.3751547038555145, "sampling/importance_sampling_ratio/mean": 0.9988890290260315, "sampling/importance_sampling_ratio/max": 1.1496003866195679, "entropy": 0.06646110257133842, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0034722222480922937, "clip_ratio/high_max": 0.0034722222480922937, "clip_ratio/region_mean": 0.0034722222480922937, "reward_total_mean": 0.9993535876274109, "reward_meter_mean": 0.9993535876274109, "reward_meter_std": 0.00013377754657994956, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993535876274109, "reward_total_composite_std": 0.00013377754657994956} {"timestamp_utc": "2026-04-12T02:04:26Z", "mode": "train", "global_step": 2443, "epoch": 0.09812427200064265, "loss": 0.1046, "grad_norm": 7.693568229675293, "learning_rate": 2.6e-06, "num_tokens": 5523328.0, "completions/mean_length": 97.375, "completions/min_length": 84.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.375, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.9511677026748657, "rewards/meter/std": 0.009777599945664406, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.12938730418682098, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.908730149269104, "rewards/repeat_penalty/std": 0.08302231132984161, "rewards/total_composite/mean": 0.729360044002533, "rewards/total_composite/std": 0.13022364675998688, "reward": 0.729360044002533, "reward_std": 0.13022364675998688, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06830313801765442, "sampling/sampling_logp_difference/max": 2.807176351547241, "sampling/importance_sampling_ratio/min": 0.06037522852420807, "sampling/importance_sampling_ratio/mean": 1.003010630607605, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3307408094406128, "clip_ratio/low_mean": 0.04029961163178086, "clip_ratio/low_min": 0.04029961163178086, "clip_ratio/high_mean": 0.026477832812815905, "clip_ratio/high_max": 0.026477832812815905, "clip_ratio/region_mean": 0.06677744444459677, "reward_total_mean": 0.729360044002533, "reward_meter_mean": 0.9511677026748657, "reward_meter_std": 0.009777599945664406, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.12938730418682098, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.908730149269104, "reward_repeat_penalty_std": 0.08302231132984161, "reward_total_composite_mean": 0.729360044002533, "reward_total_composite_std": 0.13022364675998688} {"timestamp_utc": "2026-04-12T02:04:34Z", "mode": "train", "global_step": 2444, "epoch": 0.0981644374824276, "loss": 0.0042, "grad_norm": 1.5168893337249756, "learning_rate": 2.5969696969696973e-06, "num_tokens": 5527625.0, "completions/mean_length": 320.125, "completions/min_length": 319.0, "completions/max_length": 321.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 320.125, "completions/min_terminated_length": 319.0, "completions/max_terminated_length": 321.0, "rewards/meter/mean": 0.9990435242652893, "rewards/meter/std": 0.00013361620949581265, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7426470518112183, "rewards/repeat_penalty/std": 0.062391772866249084, "rewards/total_composite/mean": 0.74193274974823, "rewards/total_composite/std": 0.0622798353433609, "reward": 0.74193274974823, "reward_std": 0.0622798316180706, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016514526680111885, "sampling/sampling_logp_difference/max": 1.009657382965088, "sampling/importance_sampling_ratio/min": 0.41537001729011536, "sampling/importance_sampling_ratio/mean": 1.005713939666748, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12365446705371141, "clip_ratio/low_mean": 0.0039014052599668503, "clip_ratio/low_min": 0.0039014052599668503, "clip_ratio/high_mean": 0.006253673695027828, "clip_ratio/high_max": 0.006253673695027828, "clip_ratio/region_mean": 0.010155078954994678, "reward_total_mean": 0.74193274974823, "reward_meter_mean": 0.9990435242652893, "reward_meter_std": 0.00013361620949581265, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7426470518112183, "reward_repeat_penalty_std": 0.062391772866249084, "reward_total_composite_mean": 0.74193274974823, "reward_total_composite_std": 0.0622798353433609} {"timestamp_utc": "2026-04-12T02:04:38Z", "mode": "train", "global_step": 2445, "epoch": 0.09820460296421256, "loss": 0.0227, "grad_norm": 8.448700904846191, "learning_rate": 2.593939393939394e-06, "num_tokens": 5529594.0, "completions/mean_length": 68.125, "completions/min_length": 65.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.125, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9891185760498047, "rewards/meter/std": 0.028624890372157097, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9891185760498047, "rewards/total_composite/std": 0.028624890372157097, "reward": 0.9891185760498047, "reward_std": 0.028624894097447395, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04746231809258461, "sampling/sampling_logp_difference/max": 1.0394535064697266, "sampling/importance_sampling_ratio/min": 0.3536478877067566, "sampling/importance_sampling_ratio/mean": 0.9979510307312012, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2945162560790777, "clip_ratio/low_mean": 0.0069444444961845875, "clip_ratio/low_min": 0.0069444444961845875, "clip_ratio/high_mean": 0.02955011453013867, "clip_ratio/high_max": 0.02955011453013867, "clip_ratio/region_mean": 0.03649455902632326, "reward_total_mean": 0.9891185760498047, "reward_meter_mean": 0.9891185760498047, "reward_meter_std": 0.028624890372157097, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9891185760498047, "reward_total_composite_std": 0.028624890372157097} {"timestamp_utc": "2026-04-12T02:04:43Z", "mode": "train", "global_step": 2446, "epoch": 0.09824476844599751, "loss": 0.0023, "grad_norm": 0.9163700938224792, "learning_rate": 2.590909090909091e-06, "num_tokens": 5531756.0, "completions/mean_length": 107.25, "completions/min_length": 106.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.25, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.9992789626121521, "rewards/meter/std": 9.53831258811988e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992789626121521, "rewards/total_composite/std": 9.53831258811988e-05, "reward": 0.9992789626121521, "reward_std": 9.538961603539065e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010050991550087929, "sampling/sampling_logp_difference/max": 0.7290101051330566, "sampling/importance_sampling_ratio/min": 0.4823862612247467, "sampling/importance_sampling_ratio/mean": 1.0028537511825562, "sampling/importance_sampling_ratio/max": 1.2234734296798706, "entropy": 0.07135625509545207, "clip_ratio/low_mean": 0.0034616037737578154, "clip_ratio/low_min": 0.0034616037737578154, "clip_ratio/high_mean": 0.003515693824738264, "clip_ratio/high_max": 0.003515693824738264, "clip_ratio/region_mean": 0.0069772975984960794, "reward_total_mean": 0.9992789626121521, "reward_meter_mean": 0.9992789626121521, "reward_meter_std": 9.53831258811988e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992789626121521, "reward_total_composite_std": 9.53831258811988e-05} {"timestamp_utc": "2026-04-12T02:04:54Z", "mode": "train", "global_step": 2447, "epoch": 0.09828493392778247, "loss": -0.0107, "grad_norm": 1.6099560260772705, "learning_rate": 2.5878787878787883e-06, "num_tokens": 5537808.0, "completions/mean_length": 447.5, "completions/min_length": 421.0, "completions/max_length": 472.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 447.5, "completions/min_terminated_length": 421.0, "completions/max_terminated_length": 472.0, "rewards/meter/mean": 0.9982467889785767, "rewards/meter/std": 0.0005631654057651758, "rewards/count_adherence/mean": 0.7583333253860474, "rewards/count_adherence/std": 0.0345032773911953, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9484989643096924, "rewards/repeat_penalty/std": 0.03878781571984291, "rewards/total_composite/mean": 0.7178632020950317, "rewards/total_composite/std": 0.040422506630420685, "reward": 0.7178632020950317, "reward_std": 0.040422506630420685, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.041198987513780594, "sampling/sampling_logp_difference/max": 1.7551989555358887, "sampling/importance_sampling_ratio/min": 0.17287284135818481, "sampling/importance_sampling_ratio/mean": 1.003982424736023, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3355287965387106, "clip_ratio/low_mean": 0.009800048777833581, "clip_ratio/low_min": 0.009800048777833581, "clip_ratio/high_mean": 0.024517936864867806, "clip_ratio/high_max": 0.024517936864867806, "clip_ratio/region_mean": 0.03431798564270139, "reward_total_mean": 0.7178632020950317, "reward_meter_mean": 0.9982467889785767, "reward_meter_std": 0.0005631654057651758, "reward_count_adherence_mean": 0.7583333253860474, "reward_count_adherence_std": 0.0345032773911953, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9484989643096924, "reward_repeat_penalty_std": 0.03878781571984291, "reward_total_composite_mean": 0.7178632020950317, "reward_total_composite_std": 0.040422506630420685} {"timestamp_utc": "2026-04-12T02:05:02Z", "mode": "train", "global_step": 2448, "epoch": 0.09832509940956742, "loss": 0.0097, "grad_norm": 1.5874656438827515, "learning_rate": 2.584848484848485e-06, "num_tokens": 5541790.0, "completions/mean_length": 314.75, "completions/min_length": 306.0, "completions/max_length": 320.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 314.75, "completions/min_terminated_length": 306.0, "completions/max_terminated_length": 320.0, "rewards/meter/mean": 0.9986426830291748, "rewards/meter/std": 0.0003120753972325474, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9333333373069763, "rewards/repeat_penalty/std": 0.07126966118812561, "rewards/total_composite/mean": 0.9320781230926514, "rewards/total_composite/std": 0.0713275596499443, "reward": 0.9320781230926514, "reward_std": 0.0713275596499443, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031515371054410934, "sampling/sampling_logp_difference/max": 1.0676155090332031, "sampling/importance_sampling_ratio/min": 0.34382739663124084, "sampling/importance_sampling_ratio/mean": 1.0063222646713257, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.26868183352053165, "clip_ratio/low_mean": 0.00553075410425663, "clip_ratio/low_min": 0.00553075410425663, "clip_ratio/high_mean": 0.01969089114572853, "clip_ratio/high_max": 0.01969089114572853, "clip_ratio/region_mean": 0.02522164524998516, "reward_total_mean": 0.9320781230926514, "reward_meter_mean": 0.9986426830291748, "reward_meter_std": 0.0003120753972325474, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9333333373069763, "reward_repeat_penalty_std": 0.07126966118812561, "reward_total_composite_mean": 0.9320781230926514, "reward_total_composite_std": 0.0713275596499443} {"timestamp_utc": "2026-04-12T02:05:09Z", "mode": "train", "global_step": 2449, "epoch": 0.09836526489135237, "loss": 0.0056, "grad_norm": 2.10903000831604, "learning_rate": 2.581818181818182e-06, "num_tokens": 5546236.0, "completions/mean_length": 314.75, "completions/min_length": 313.0, "completions/max_length": 315.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 314.75, "completions/min_terminated_length": 313.0, "completions/max_terminated_length": 315.0, "rewards/meter/mean": 0.9972984790802002, "rewards/meter/std": 7.734992686891928e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6907894611358643, "rewards/repeat_penalty/std": 0.06560124456882477, "rewards/total_composite/mean": 0.6889257431030273, "rewards/total_composite/std": 0.0654601976275444, "reward": 0.6889257431030273, "reward_std": 0.06546018272638321, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014586514793336391, "sampling/sampling_logp_difference/max": 1.952465534210205, "sampling/importance_sampling_ratio/min": 0.141923725605011, "sampling/importance_sampling_ratio/mean": 1.0018553733825684, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07688481360673904, "clip_ratio/low_mean": 0.0027777778159361333, "clip_ratio/low_min": 0.0027777778159361333, "clip_ratio/high_mean": 0.002396166091784835, "clip_ratio/high_max": 0.002396166091784835, "clip_ratio/region_mean": 0.005173943907720968, "reward_total_mean": 0.6889257431030273, "reward_meter_mean": 0.9972984790802002, "reward_meter_std": 7.734992686891928e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6907894611358643, "reward_repeat_penalty_std": 0.06560124456882477, "reward_total_composite_mean": 0.6889257431030273, "reward_total_composite_std": 0.0654601976275444} {"timestamp_utc": "2026-04-12T02:05:14Z", "mode": "train", "global_step": 2450, "epoch": 0.09840543037313733, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.5787878787878788e-06, "num_tokens": 5547852.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0006953292759135365, "sampling/sampling_logp_difference/max": 0.06802265346050262, "sampling/importance_sampling_ratio/min": 0.9342393279075623, "sampling/importance_sampling_ratio/mean": 1.000256061553955, "sampling/importance_sampling_ratio/max": 1.0135822296142578, "entropy": 0.004708475491497666, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:06:29Z", "mode": "eval", "global_step": 2450, "epoch": 0.09840543037313733, "eval_loss": NaN, "eval_runtime": 74.6804, "eval_samples_per_second": 1.393, "eval_steps_per_second": 0.174, "eval_num_tokens": 5547852.0, "eval_completions/mean_length": 203.57692307692307, "eval_completions/min_length": 61.38461538461539, "eval_completions/max_length": 394.38461538461536, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 203.57692307692307, "eval_completions/min_terminated_length": 61.38461538461539, "eval_completions/max_terminated_length": 394.38461538461536, "eval_rewards/meter/mean": 0.7445399669500498, "eval_rewards/meter/std": 0.3695275445397084, "eval_rewards/count_adherence/mean": 0.9357927395747259, "eval_rewards/count_adherence/std": 0.08574442823345844, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/repeat_penalty/mean": 0.8948483146153964, "eval_rewards/repeat_penalty/std": 0.11098527220579293, "eval_rewards/total_composite/mean": 0.6300999980706435, "eval_rewards/total_composite/std": 0.3479128239246515, "eval_reward": 0.6300999980706435, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.022716052543658476, "eval_sampling/sampling_logp_difference/max": 0.9992542266845703, "eval_sampling/importance_sampling_ratio/min": 0.37165725231170654, "eval_sampling/importance_sampling_ratio/mean": 1.0057257138765776, "eval_sampling/importance_sampling_ratio/max": 1.4525978748614972, "eval_entropy": 0.23695378464001876, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6300999980706435, "eval_reward_meter_mean": 0.7445399669500498, "eval_reward_meter_std": 0.3695275445397084, "eval_reward_count_adherence_mean": 0.9357927395747259, "eval_reward_count_adherence_std": 0.08574442823345844, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_repeat_penalty_mean": 0.8948483146153964, "eval_reward_repeat_penalty_std": 0.11098527220579293, "eval_reward_total_composite_mean": 0.6300999980706435, "eval_reward_total_composite_std": 0.3479128239246515} {"timestamp_utc": "2026-04-12T02:06:35Z", "mode": "train", "global_step": 2451, "epoch": 0.09844559585492228, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.575757575757576e-06, "num_tokens": 5549564.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0012545905774459243, "sampling/sampling_logp_difference/max": 0.19019845128059387, "sampling/importance_sampling_ratio/min": 0.8267950415611267, "sampling/importance_sampling_ratio/mean": 1.000001311302185, "sampling/importance_sampling_ratio/max": 1.0180944204330444, "entropy": 0.005932128522545099, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:06:40Z", "mode": "train", "global_step": 2452, "epoch": 0.09848576133670724, "loss": 0.1072, "grad_norm": 8.379902839660645, "learning_rate": 2.572727272727273e-06, "num_tokens": 5551659.0, "completions/mean_length": 92.875, "completions/min_length": 85.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.875, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.9926849007606506, "rewards/meter/std": 0.0018546542851254344, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9321428537368774, "rewards/repeat_penalty/std": 0.09529759734869003, "rewards/total_composite/mean": 0.8898848295211792, "rewards/total_composite/std": 0.1583697348833084, "reward": 0.8898848295211792, "reward_std": 0.15836970508098602, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.038993846625089645, "sampling/sampling_logp_difference/max": 1.2752385139465332, "sampling/importance_sampling_ratio/min": 0.27936431765556335, "sampling/importance_sampling_ratio/mean": 1.0090692043304443, "sampling/importance_sampling_ratio/max": 1.9530335664749146, "entropy": 0.2788791488856077, "clip_ratio/low_mean": 0.013440471375361085, "clip_ratio/low_min": 0.013440471375361085, "clip_ratio/high_mean": 0.01974922022782266, "clip_ratio/high_max": 0.01974922022782266, "clip_ratio/region_mean": 0.033189691603183746, "reward_total_mean": 0.8898848295211792, "reward_meter_mean": 0.9926849007606506, "reward_meter_std": 0.0018546542851254344, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9321428537368774, "reward_repeat_penalty_std": 0.09529759734869003, "reward_total_composite_mean": 0.8898848295211792, "reward_total_composite_std": 0.1583697348833084} {"timestamp_utc": "2026-04-12T02:06:45Z", "mode": "train", "global_step": 2453, "epoch": 0.09852592681849219, "loss": -0.0104, "grad_norm": 4.428008079528809, "learning_rate": 2.5696969696969697e-06, "num_tokens": 5553819.0, "completions/mean_length": 89.0, "completions/min_length": 84.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.0, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.9919699430465698, "rewards/meter/std": 0.005432384088635445, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9670881032943726, "rewards/total_composite/std": 0.06924090534448624, "reward": 0.9670881032943726, "reward_std": 0.06924089044332504, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031983181834220886, "sampling/sampling_logp_difference/max": 1.074939250946045, "sampling/importance_sampling_ratio/min": 0.35889124870300293, "sampling/importance_sampling_ratio/mean": 1.0001150369644165, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.21917428076267242, "clip_ratio/low_mean": 0.004360465332865715, "clip_ratio/low_min": 0.004360465332865715, "clip_ratio/high_mean": 0.036513578495942056, "clip_ratio/high_max": 0.036513578495942056, "clip_ratio/region_mean": 0.04087404382880777, "reward_total_mean": 0.9670881032943726, "reward_meter_mean": 0.9919699430465698, "reward_meter_std": 0.005432384088635445, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9670881032943726, "reward_total_composite_std": 0.06924090534448624} {"timestamp_utc": "2026-04-12T02:06:50Z", "mode": "train", "global_step": 2454, "epoch": 0.09856609230027714, "loss": 0.0216, "grad_norm": 4.521707534790039, "learning_rate": 2.566666666666667e-06, "num_tokens": 5555436.0, "completions/mean_length": 40.125, "completions/min_length": 38.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.125, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.9973416328430176, "rewards/meter/std": 0.0012014260282739997, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973416328430176, "rewards/total_composite/std": 0.0012014260282739997, "reward": 0.9973416328430176, "reward_std": 0.0012014324311167002, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017517054453492165, "sampling/sampling_logp_difference/max": 1.4276123046875, "sampling/importance_sampling_ratio/min": 0.2398810088634491, "sampling/importance_sampling_ratio/mean": 0.9969314932823181, "sampling/importance_sampling_ratio/max": 1.7687947750091553, "entropy": 0.11128469556570053, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0030487803742289543, "clip_ratio/high_max": 0.0030487803742289543, "clip_ratio/region_mean": 0.0030487803742289543, "reward_total_mean": 0.9973416328430176, "reward_meter_mean": 0.9973416328430176, "reward_meter_std": 0.0012014260282739997, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9973416328430176, "reward_total_composite_std": 0.0012014260282739997} {"timestamp_utc": "2026-04-12T02:06:54Z", "mode": "train", "global_step": 2455, "epoch": 0.0986062577820621, "loss": 0.2385, "grad_norm": 2.2290995121002197, "learning_rate": 2.5636363636363638e-06, "num_tokens": 5556947.0, "completions/mean_length": 36.875, "completions/min_length": 33.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.875, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9920687675476074, "rewards/meter/std": 0.00039521773578599095, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8681668043136597, "rewards/total_composite/std": 0.35079243779182434, "reward": 0.8681668043136597, "reward_std": 0.35079240798950195, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016091084107756615, "sampling/sampling_logp_difference/max": 1.2472076416015625, "sampling/importance_sampling_ratio/min": 0.28730595111846924, "sampling/importance_sampling_ratio/mean": 1.0059282779693604, "sampling/importance_sampling_ratio/max": 1.3825856447219849, "entropy": 0.11671794764697552, "clip_ratio/low_mean": 0.002016128972172737, "clip_ratio/low_min": 0.002016128972172737, "clip_ratio/high_mean": 0.007464349502697587, "clip_ratio/high_max": 0.007464349502697587, "clip_ratio/region_mean": 0.009480478474870324, "reward_total_mean": 0.8681668043136597, "reward_meter_mean": 0.9920687675476074, "reward_meter_std": 0.00039521773578599095, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8681668043136597, "reward_total_composite_std": 0.35079243779182434} {"timestamp_utc": "2026-04-12T02:06:59Z", "mode": "train", "global_step": 2456, "epoch": 0.09864642326384705, "loss": 0.0052, "grad_norm": 3.4946515560150146, "learning_rate": 2.5606060606060606e-06, "num_tokens": 5558871.0, "completions/mean_length": 80.5, "completions/min_length": 79.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.5, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.9981932044029236, "rewards/meter/std": 0.0004672010545618832, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981932044029236, "rewards/total_composite/std": 0.0004672010545618832, "reward": 0.9981932044029236, "reward_std": 0.0004671814094763249, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02568862773478031, "sampling/sampling_logp_difference/max": 1.0305986404418945, "sampling/importance_sampling_ratio/min": 0.3567933142185211, "sampling/importance_sampling_ratio/mean": 1.0071548223495483, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.22475284524261951, "clip_ratio/low_mean": 0.003088302561081946, "clip_ratio/low_min": 0.003088302561081946, "clip_ratio/high_mean": 0.007773919845931232, "clip_ratio/high_max": 0.007773919845931232, "clip_ratio/region_mean": 0.010862222407013178, "reward_total_mean": 0.9981932044029236, "reward_meter_mean": 0.9981932044029236, "reward_meter_std": 0.0004672010545618832, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981932044029236, "reward_total_composite_std": 0.0004672010545618832} {"timestamp_utc": "2026-04-12T02:07:05Z", "mode": "train", "global_step": 2457, "epoch": 0.098686588745632, "loss": 0.0114, "grad_norm": 2.7975685596466064, "learning_rate": 2.5575757575757574e-06, "num_tokens": 5561940.0, "completions/mean_length": 183.625, "completions/min_length": 177.0, "completions/max_length": 190.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 183.625, "completions/min_terminated_length": 177.0, "completions/max_terminated_length": 190.0, "rewards/meter/mean": 0.9926738142967224, "rewards/meter/std": 0.000535113038495183, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9318182468414307, "rewards/repeat_penalty/std": 0.04208271950483322, "rewards/total_composite/mean": 0.9250072836875916, "rewards/total_composite/std": 0.042194660753011703, "reward": 0.9250072836875916, "reward_std": 0.042194657027721405, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03491319715976715, "sampling/sampling_logp_difference/max": 3.0987467765808105, "sampling/importance_sampling_ratio/min": 0.045105695724487305, "sampling/importance_sampling_ratio/mean": 0.9990267157554626, "sampling/importance_sampling_ratio/max": 1.8111183643341064, "entropy": 0.2139601055532694, "clip_ratio/low_mean": 0.01427258289186284, "clip_ratio/low_min": 0.01427258289186284, "clip_ratio/high_mean": 0.00900993263348937, "clip_ratio/high_max": 0.00900993263348937, "clip_ratio/region_mean": 0.02328251552535221, "reward_total_mean": 0.9250072836875916, "reward_meter_mean": 0.9926738142967224, "reward_meter_std": 0.000535113038495183, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9318182468414307, "reward_repeat_penalty_std": 0.04208271950483322, "reward_total_composite_mean": 0.9250072836875916, "reward_total_composite_std": 0.042194660753011703} {"timestamp_utc": "2026-04-12T02:07:09Z", "mode": "train", "global_step": 2458, "epoch": 0.09872675422741696, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.5545454545454547e-06, "num_tokens": 5563996.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00065056630410254, "sampling/sampling_logp_difference/max": 0.08942283689975739, "sampling/importance_sampling_ratio/min": 0.9144588112831116, "sampling/importance_sampling_ratio/mean": 1.0000568628311157, "sampling/importance_sampling_ratio/max": 1.0292819738388062, "entropy": 0.005484737688675523, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:07:15Z", "mode": "train", "global_step": 2459, "epoch": 0.09876691970920191, "loss": -0.0206, "grad_norm": 6.4525146484375, "learning_rate": 2.5515151515151515e-06, "num_tokens": 5566684.0, "completions/mean_length": 164.0, "completions/min_length": 143.0, "completions/max_length": 170.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 164.0, "completions/min_terminated_length": 143.0, "completions/max_terminated_length": 170.0, "rewards/meter/mean": 0.9948143362998962, "rewards/meter/std": 0.010059010237455368, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.9674677848815918, "rewards/total_composite/std": 0.05743533372879028, "reward": 0.9674677848815918, "reward_std": 0.057435326278209686, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.054715219885110855, "sampling/sampling_logp_difference/max": 1.7331323623657227, "sampling/importance_sampling_ratio/min": 0.1767299622297287, "sampling/importance_sampling_ratio/mean": 1.0046672821044922, "sampling/importance_sampling_ratio/max": 1.882559061050415, "entropy": 0.37010542675852776, "clip_ratio/low_mean": 0.009364139288663864, "clip_ratio/low_min": 0.009364139288663864, "clip_ratio/high_mean": 0.03573530330322683, "clip_ratio/high_max": 0.03573530330322683, "clip_ratio/region_mean": 0.04509944259189069, "reward_total_mean": 0.9674677848815918, "reward_meter_mean": 0.9948143362998962, "reward_meter_std": 0.010059010237455368, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_total_composite_mean": 0.9674677848815918, "reward_total_composite_std": 0.05743533372879028} {"timestamp_utc": "2026-04-12T02:07:20Z", "mode": "train", "global_step": 2460, "epoch": 0.09880708519098687, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.5484848484848484e-06, "num_tokens": 5568492.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002486900775693357, "sampling/sampling_logp_difference/max": 0.013540960848331451, "sampling/importance_sampling_ratio/min": 0.9955151677131653, "sampling/importance_sampling_ratio/mean": 1.0002198219299316, "sampling/importance_sampling_ratio/max": 1.0136330127716064, "entropy": 0.0032962941040750593, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:07:29Z", "mode": "train", "global_step": 2461, "epoch": 0.09884725067277182, "loss": -0.023, "grad_norm": 1.6912506818771362, "learning_rate": 2.5454545454545456e-06, "num_tokens": 5573680.0, "completions/mean_length": 404.5, "completions/min_length": 378.0, "completions/max_length": 424.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 404.5, "completions/min_terminated_length": 378.0, "completions/max_terminated_length": 424.0, "rewards/meter/mean": 0.9983578324317932, "rewards/meter/std": 0.0009692087187431753, "rewards/count_adherence/mean": 0.7980769276618958, "rewards/count_adherence/std": 0.05723259598016739, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9470028877258301, "rewards/repeat_penalty/std": 0.06291526556015015, "rewards/total_composite/mean": 0.7565553188323975, "rewards/total_composite/std": 0.09492313861846924, "reward": 0.7565553188323975, "reward_std": 0.09492312371730804, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.042006395757198334, "sampling/sampling_logp_difference/max": 1.7684879302978516, "sampling/importance_sampling_ratio/min": 0.17059072852134705, "sampling/importance_sampling_ratio/mean": 1.0055227279663086, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.35317912697792053, "clip_ratio/low_mean": 0.008652225020341575, "clip_ratio/low_min": 0.008652225020341575, "clip_ratio/high_mean": 0.022989682853221893, "clip_ratio/high_max": 0.022989682853221893, "clip_ratio/region_mean": 0.03164190787356347, "reward_total_mean": 0.7565553188323975, "reward_meter_mean": 0.9983578324317932, "reward_meter_std": 0.0009692087187431753, "reward_count_adherence_mean": 0.7980769276618958, "reward_count_adherence_std": 0.05723259598016739, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9470028877258301, "reward_repeat_penalty_std": 0.06291526556015015, "reward_total_composite_mean": 0.7565553188323975, "reward_total_composite_std": 0.09492313861846924} {"timestamp_utc": "2026-04-12T02:07:34Z", "mode": "train", "global_step": 2462, "epoch": 0.09888741615455678, "loss": 0.0004, "grad_norm": 0.51884925365448, "learning_rate": 2.542424242424243e-06, "num_tokens": 5575497.0, "completions/mean_length": 67.125, "completions/min_length": 67.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.125, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9981344938278198, "rewards/meter/std": 2.5179730073432438e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981344938278198, "rewards/total_composite/std": 2.5179730073432438e-05, "reward": 0.9981344938278198, "reward_std": 2.518050132493954e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007205577101558447, "sampling/sampling_logp_difference/max": 0.981438159942627, "sampling/importance_sampling_ratio/min": 0.37477174401283264, "sampling/importance_sampling_ratio/mean": 0.9990748167037964, "sampling/importance_sampling_ratio/max": 1.997531771659851, "entropy": 0.017152427230030298, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/high_mean": 0.0037313431967049837, "clip_ratio/high_max": 0.0037313431967049837, "clip_ratio/region_mean": 0.005597014795057476, "reward_total_mean": 0.9981344938278198, "reward_meter_mean": 0.9981344938278198, "reward_meter_std": 2.5179730073432438e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981344938278198, "reward_total_composite_std": 2.5179730073432438e-05} {"timestamp_utc": "2026-04-12T02:07:39Z", "mode": "train", "global_step": 2463, "epoch": 0.09892758163634173, "loss": 0.0004, "grad_norm": 0.14106905460357666, "learning_rate": 2.5393939393939397e-06, "num_tokens": 5577697.0, "completions/mean_length": 98.0, "completions/min_length": 98.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.0, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9980115294456482, "rewards/meter/std": 1.5354547940660268e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980115294456482, "rewards/total_composite/std": 1.5354547940660268e-05, "reward": 0.9980115294456482, "reward_std": 1.5354515198851004e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0028446686919778585, "sampling/sampling_logp_difference/max": 0.4193582534790039, "sampling/importance_sampling_ratio/min": 0.6574686169624329, "sampling/importance_sampling_ratio/mean": 1.0014915466308594, "sampling/importance_sampling_ratio/max": 1.2301760911941528, "entropy": 0.027404201216995716, "clip_ratio/low_mean": 0.0012755101779475808, "clip_ratio/low_min": 0.0012755101779475808, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0012755101779475808, "reward_total_mean": 0.9980115294456482, "reward_meter_mean": 0.9980115294456482, "reward_meter_std": 1.5354547940660268e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9980115294456482, "reward_total_composite_std": 1.5354547940660268e-05} {"timestamp_utc": "2026-04-12T02:07:43Z", "mode": "train", "global_step": 2464, "epoch": 0.09896774711812668, "loss": -0.0148, "grad_norm": 2.84199595451355, "learning_rate": 2.536363636363637e-06, "num_tokens": 5579505.0, "completions/mean_length": 61.0, "completions/min_length": 58.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9931963086128235, "rewards/meter/std": 0.00066983891883865, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9931963086128235, "rewards/total_composite/std": 0.00066983891883865, "reward": 0.9931963086128235, "reward_std": 0.0006698234938085079, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013205956667661667, "sampling/sampling_logp_difference/max": 0.7910587787628174, "sampling/importance_sampling_ratio/min": 0.4533645212650299, "sampling/importance_sampling_ratio/mean": 1.0031862258911133, "sampling/importance_sampling_ratio/max": 1.462441325187683, "entropy": 0.09446914121508598, "clip_ratio/low_mean": 0.008408705238252878, "clip_ratio/low_min": 0.008408705238252878, "clip_ratio/high_mean": 0.004032257944345474, "clip_ratio/high_max": 0.004032257944345474, "clip_ratio/region_mean": 0.012440963182598352, "reward_total_mean": 0.9931963086128235, "reward_meter_mean": 0.9931963086128235, "reward_meter_std": 0.00066983891883865, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9931963086128235, "reward_total_composite_std": 0.00066983891883865} {"timestamp_utc": "2026-04-12T02:07:51Z", "mode": "train", "global_step": 2465, "epoch": 0.09900791259991164, "loss": 0.0112, "grad_norm": 1.982505440711975, "learning_rate": 2.5333333333333338e-06, "num_tokens": 5583250.0, "completions/mean_length": 271.125, "completions/min_length": 264.0, "completions/max_length": 279.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 271.125, "completions/min_terminated_length": 264.0, "completions/max_terminated_length": 279.0, "rewards/meter/mean": 0.998477578163147, "rewards/meter/std": 0.0006886826595291495, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9711538553237915, "rewards/repeat_penalty/std": 0.05723259598016739, "rewards/total_composite/mean": 0.969672441482544, "rewards/total_composite/std": 0.057110171765089035, "reward": 0.969672441482544, "reward_std": 0.05711015686392784, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.043706730008125305, "sampling/sampling_logp_difference/max": 1.9009828567504883, "sampling/importance_sampling_ratio/min": 0.14942169189453125, "sampling/importance_sampling_ratio/mean": 1.0076007843017578, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39818970672786236, "clip_ratio/low_mean": 0.006513258791528642, "clip_ratio/low_min": 0.006513258791528642, "clip_ratio/high_mean": 0.028696169378235936, "clip_ratio/high_max": 0.028696169378235936, "clip_ratio/region_mean": 0.03520942816976458, "reward_total_mean": 0.969672441482544, "reward_meter_mean": 0.998477578163147, "reward_meter_std": 0.0006886826595291495, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9711538553237915, "reward_repeat_penalty_std": 0.05723259598016739, "reward_total_composite_mean": 0.969672441482544, "reward_total_composite_std": 0.057110171765089035} {"timestamp_utc": "2026-04-12T02:07:56Z", "mode": "train", "global_step": 2466, "epoch": 0.09904807808169659, "loss": 0.0039, "grad_norm": 1.754108190536499, "learning_rate": 2.5303030303030306e-06, "num_tokens": 5585974.0, "completions/mean_length": 142.5, "completions/min_length": 142.0, "completions/max_length": 143.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 142.5, "completions/min_terminated_length": 142.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.9990931153297424, "rewards/meter/std": 0.00018217807519249618, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.910714328289032, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9098852872848511, "rewards/total_composite/std": 0.07382422685623169, "reward": 0.9098852872848511, "reward_std": 0.07382424175739288, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014981756918132305, "sampling/sampling_logp_difference/max": 0.8543601036071777, "sampling/importance_sampling_ratio/min": 0.42555540800094604, "sampling/importance_sampling_ratio/mean": 1.005768060684204, "sampling/importance_sampling_ratio/max": 1.6116867065429688, "entropy": 0.095588818192482, "clip_ratio/low_mean": 0.010514133609831333, "clip_ratio/low_min": 0.010514133609831333, "clip_ratio/high_mean": 0.004401408368721604, "clip_ratio/high_max": 0.004401408368721604, "clip_ratio/region_mean": 0.014915541978552938, "reward_total_mean": 0.9098852872848511, "reward_meter_mean": 0.9990931153297424, "reward_meter_std": 0.00018217807519249618, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.910714328289032, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.9098852872848511, "reward_total_composite_std": 0.07382422685623169} {"timestamp_utc": "2026-04-12T02:08:00Z", "mode": "train", "global_step": 2467, "epoch": 0.09908824356348155, "loss": 0.0003, "grad_norm": 0.4351802468299866, "learning_rate": 2.5272727272727274e-06, "num_tokens": 5587830.0, "completions/mean_length": 80.0, "completions/min_length": 80.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.0, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.7115719318389893, "rewards/meter/std": 0.021074457094073296, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.6950865983963013, "rewards/total_composite/std": 0.06770216673612595, "reward": 0.6950865983963013, "reward_std": 0.06770216673612595, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0018030240898951888, "sampling/sampling_logp_difference/max": 0.2816805839538574, "sampling/importance_sampling_ratio/min": 0.8967838287353516, "sampling/importance_sampling_ratio/mean": 1.0006452798843384, "sampling/importance_sampling_ratio/max": 1.3253552913665771, "entropy": 0.013048171065747738, "clip_ratio/low_mean": 0.0015625000232830644, "clip_ratio/low_min": 0.0015625000232830644, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0015625000232830644, "reward_total_mean": 0.6950865983963013, "reward_meter_mean": 0.7115719318389893, "reward_meter_std": 0.021074457094073296, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.6950865983963013, "reward_total_composite_std": 0.06770216673612595} {"timestamp_utc": "2026-04-12T02:08:05Z", "mode": "train", "global_step": 2468, "epoch": 0.0991284090452665, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.5242424242424247e-06, "num_tokens": 5589294.0, "completions/mean_length": 32.0, "completions/min_length": 32.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "reward": 0.9992982149124146, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00015046881162561476, "sampling/sampling_logp_difference/max": 0.0030941502191126347, "sampling/importance_sampling_ratio/min": 0.9973318576812744, "sampling/importance_sampling_ratio/mean": 1.0000958442687988, "sampling/importance_sampling_ratio/max": 1.003098964691162, "entropy": 0.0013736404580413364, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9992982149124146, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:08:09Z", "mode": "train", "global_step": 2469, "epoch": 0.09916857452705145, "loss": -0.0232, "grad_norm": 4.22209358215332, "learning_rate": 2.5212121212121215e-06, "num_tokens": 5591249.0, "completions/mean_length": 90.375, "completions/min_length": 84.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.375, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.9939854145050049, "rewards/meter/std": 0.0009144411887973547, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9691168665885925, "rewards/total_composite/std": 0.07003802061080933, "reward": 0.9691168665885925, "reward_std": 0.07003801316022873, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022249562665820122, "sampling/sampling_logp_difference/max": 1.9335792064666748, "sampling/importance_sampling_ratio/min": 0.1446296125650406, "sampling/importance_sampling_ratio/mean": 0.9984654784202576, "sampling/importance_sampling_ratio/max": 1.7310431003570557, "entropy": 0.08473077462986112, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.023276447434909642, "clip_ratio/high_max": 0.023276447434909642, "clip_ratio/region_mean": 0.023276447434909642, "reward_total_mean": 0.9691168665885925, "reward_meter_mean": 0.9939854145050049, "reward_meter_std": 0.0009144411887973547, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9691168665885925, "reward_total_composite_std": 0.07003802061080933} {"timestamp_utc": "2026-04-12T02:08:15Z", "mode": "train", "global_step": 2470, "epoch": 0.09920874000883641, "loss": -0.0038, "grad_norm": 2.6985630989074707, "learning_rate": 2.5181818181818184e-06, "num_tokens": 5594095.0, "completions/mean_length": 163.75, "completions/min_length": 159.0, "completions/max_length": 168.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 163.75, "completions/min_terminated_length": 159.0, "completions/max_terminated_length": 168.0, "rewards/meter/mean": 0.998643159866333, "rewards/meter/std": 0.0008321683271788061, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.9708901643753052, "rewards/total_composite/std": 0.0511077418923378, "reward": 0.9708901643753052, "reward_std": 0.05110771581530571, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.049378104507923126, "sampling/sampling_logp_difference/max": 1.4132513999938965, "sampling/importance_sampling_ratio/min": 0.24335075914859772, "sampling/importance_sampling_ratio/mean": 1.012289047241211, "sampling/importance_sampling_ratio/max": 1.9667812585830688, "entropy": 0.4189731851220131, "clip_ratio/low_mean": 0.010085079353302717, "clip_ratio/low_min": 0.010085079353302717, "clip_ratio/high_mean": 0.03505220194347203, "clip_ratio/high_max": 0.03505220194347203, "clip_ratio/region_mean": 0.045137281296774745, "reward_total_mean": 0.9708901643753052, "reward_meter_mean": 0.998643159866333, "reward_meter_std": 0.0008321683271788061, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_total_composite_mean": 0.9708901643753052, "reward_total_composite_std": 0.0511077418923378} {"timestamp_utc": "2026-04-12T02:08:19Z", "mode": "train", "global_step": 2471, "epoch": 0.09924890549062136, "loss": -0.0009, "grad_norm": 1.5141384601593018, "learning_rate": 2.5151515151515156e-06, "num_tokens": 5595647.0, "completions/mean_length": 35.0, "completions/min_length": 35.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9994994401931763, "rewards/meter/std": 5.432292527984828e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994994401931763, "rewards/total_composite/std": 5.432292527984828e-05, "reward": 0.9994994401931763, "reward_std": 5.4318112233886495e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019814442843198776, "sampling/sampling_logp_difference/max": 1.097601056098938, "sampling/importance_sampling_ratio/min": 0.33367058634757996, "sampling/importance_sampling_ratio/mean": 1.0052924156188965, "sampling/importance_sampling_ratio/max": 1.4859408140182495, "entropy": 0.1270341333001852, "clip_ratio/low_mean": 0.010714285774156451, "clip_ratio/low_min": 0.010714285774156451, "clip_ratio/high_mean": 0.0035714285913854837, "clip_ratio/high_max": 0.0035714285913854837, "clip_ratio/region_mean": 0.014285714365541935, "reward_total_mean": 0.9994994401931763, "reward_meter_mean": 0.9994994401931763, "reward_meter_std": 5.432292527984828e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994994401931763, "reward_total_composite_std": 5.432292527984828e-05} {"timestamp_utc": "2026-04-12T02:08:24Z", "mode": "train", "global_step": 2472, "epoch": 0.09928907097240632, "loss": -0.0036, "grad_norm": 3.565403699874878, "learning_rate": 2.5121212121212125e-06, "num_tokens": 5598009.0, "completions/mean_length": 105.25, "completions/min_length": 102.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.25, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.45625758171081543, "rewards/meter/std": 0.3081152141094208, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.38120073080062866, "rewards/total_composite/std": 0.2558993995189667, "reward": 0.38120073080062866, "reward_std": 0.2558993995189667, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005791721399873495, "sampling/sampling_logp_difference/max": 0.5239999294281006, "sampling/importance_sampling_ratio/min": 0.5921472907066345, "sampling/importance_sampling_ratio/mean": 1.0014318227767944, "sampling/importance_sampling_ratio/max": 1.5240379571914673, "entropy": 0.025531375431455672, "clip_ratio/low_mean": 0.006092437193728983, "clip_ratio/low_min": 0.006092437193728983, "clip_ratio/high_mean": 0.002358490601181984, "clip_ratio/high_max": 0.002358490601181984, "clip_ratio/region_mean": 0.008450927794910967, "reward_total_mean": 0.38120073080062866, "reward_meter_mean": 0.45625758171081543, "reward_meter_std": 0.3081152141094208, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.07636035233736038, "reward_total_composite_mean": 0.38120073080062866, "reward_total_composite_std": 0.2558993995189667} {"timestamp_utc": "2026-04-12T02:08:28Z", "mode": "train", "global_step": 2473, "epoch": 0.09932923645419127, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.5090909090909093e-06, "num_tokens": 5599793.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002734074951149523, "sampling/sampling_logp_difference/max": 0.016472017392516136, "sampling/importance_sampling_ratio/min": 0.9836629033088684, "sampling/importance_sampling_ratio/mean": 1.0000567436218262, "sampling/importance_sampling_ratio/max": 1.010071873664856, "entropy": 0.0028824398177675903, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:08:35Z", "mode": "train", "global_step": 2474, "epoch": 0.09936940193597622, "loss": 0.0038, "grad_norm": 1.5029007196426392, "learning_rate": 2.506060606060606e-06, "num_tokens": 5603217.0, "completions/mean_length": 221.0, "completions/min_length": 218.0, "completions/max_length": 223.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 221.0, "completions/min_terminated_length": 218.0, "completions/max_terminated_length": 223.0, "rewards/meter/mean": 0.9701865911483765, "rewards/meter/std": 0.02245534211397171, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7232142686843872, "rewards/repeat_penalty/std": 0.05960877984762192, "rewards/total_composite/mean": 0.7016157507896423, "rewards/total_composite/std": 0.06003959849476814, "reward": 0.7016157507896423, "reward_std": 0.060039594769477844, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02541784755885601, "sampling/sampling_logp_difference/max": 5.142219543457031, "sampling/importance_sampling_ratio/min": 0.0058447024784982204, "sampling/importance_sampling_ratio/mean": 0.9988224506378174, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.079356012865901, "clip_ratio/low_mean": 0.015850404044613242, "clip_ratio/low_min": 0.015850404044613242, "clip_ratio/high_mean": 0.003954203391913325, "clip_ratio/high_max": 0.003954203391913325, "clip_ratio/region_mean": 0.019804607436526567, "reward_total_mean": 0.7016157507896423, "reward_meter_mean": 0.9701865911483765, "reward_meter_std": 0.02245534211397171, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7232142686843872, "reward_repeat_penalty_std": 0.05960877984762192, "reward_total_composite_mean": 0.7016157507896423, "reward_total_composite_std": 0.06003959849476814} {"timestamp_utc": "2026-04-12T02:08:41Z", "mode": "train", "global_step": 2475, "epoch": 0.09940956741776118, "loss": 0.0005, "grad_norm": 1.4216914176940918, "learning_rate": 2.5030303030303034e-06, "num_tokens": 5606940.0, "completions/mean_length": 246.375, "completions/min_length": 244.0, "completions/max_length": 248.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 246.375, "completions/min_terminated_length": 244.0, "completions/max_terminated_length": 248.0, "rewards/meter/mean": 0.9980918169021606, "rewards/meter/std": 0.00017555728845763952, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7211538553237915, "rewards/repeat_penalty/std": 0.1158415824174881, "rewards/total_composite/mean": 0.7197900414466858, "rewards/total_composite/std": 0.11570559442043304, "reward": 0.7197900414466858, "reward_std": 0.11570560187101364, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020571939647197723, "sampling/sampling_logp_difference/max": 1.042820930480957, "sampling/importance_sampling_ratio/min": 0.35245901346206665, "sampling/importance_sampling_ratio/mean": 1.004142165184021, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.14181741513311863, "clip_ratio/low_mean": 0.006118385819718242, "clip_ratio/low_min": 0.006118385819718242, "clip_ratio/high_mean": 0.010605149669572711, "clip_ratio/high_max": 0.010605149669572711, "clip_ratio/region_mean": 0.016723535489290953, "reward_total_mean": 0.7197900414466858, "reward_meter_mean": 0.9980918169021606, "reward_meter_std": 0.00017555728845763952, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7211538553237915, "reward_repeat_penalty_std": 0.1158415824174881, "reward_total_composite_mean": 0.7197900414466858, "reward_total_composite_std": 0.11570559442043304} {"timestamp_utc": "2026-04-12T02:08:46Z", "mode": "train", "global_step": 2476, "epoch": 0.09944973289954613, "loss": -0.003, "grad_norm": 3.1809544563293457, "learning_rate": 2.5e-06, "num_tokens": 5608790.0, "completions/mean_length": 68.25, "completions/min_length": 67.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.25, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9992682933807373, "rewards/meter/std": 0.00015381410776171833, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992682933807373, "rewards/total_composite/std": 0.00015381410776171833, "reward": 0.9992682933807373, "reward_std": 0.00015381253615487367, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03672777861356735, "sampling/sampling_logp_difference/max": 0.9773459434509277, "sampling/importance_sampling_ratio/min": 0.3855144679546356, "sampling/importance_sampling_ratio/mean": 1.0070955753326416, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3049812354147434, "clip_ratio/low_mean": 0.014760755351744592, "clip_ratio/low_min": 0.014760755351744592, "clip_ratio/high_mean": 0.02171556802932173, "clip_ratio/high_max": 0.02171556802932173, "clip_ratio/region_mean": 0.03647632338106632, "reward_total_mean": 0.9992682933807373, "reward_meter_mean": 0.9992682933807373, "reward_meter_std": 0.00015381410776171833, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992682933807373, "reward_total_composite_std": 0.00015381410776171833} {"timestamp_utc": "2026-04-12T02:08:54Z", "mode": "train", "global_step": 2477, "epoch": 0.09948989838133108, "loss": 0.001, "grad_norm": 2.115460157394409, "learning_rate": 2.496969696969697e-06, "num_tokens": 5613495.0, "completions/mean_length": 361.125, "completions/min_length": 339.0, "completions/max_length": 372.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 361.125, "completions/min_terminated_length": 339.0, "completions/max_terminated_length": 372.0, "rewards/meter/mean": 0.9980402588844299, "rewards/meter/std": 0.00200492306612432, "rewards/count_adherence/mean": 0.7857142686843872, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.910714328289032, "rewards/repeat_penalty/std": 0.06458108127117157, "rewards/total_composite/mean": 0.714102029800415, "rewards/total_composite/std": 0.04969966039061546, "reward": 0.714102029800415, "reward_std": 0.049699652940034866, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.047193821519613266, "sampling/sampling_logp_difference/max": 1.9314990043640137, "sampling/importance_sampling_ratio/min": 0.14493077993392944, "sampling/importance_sampling_ratio/mean": 1.0080512762069702, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3918083682656288, "clip_ratio/low_mean": 0.015457271714694798, "clip_ratio/low_min": 0.015457271714694798, "clip_ratio/high_mean": 0.02343087922781706, "clip_ratio/high_max": 0.02343087922781706, "clip_ratio/region_mean": 0.03888815094251186, "reward_total_mean": 0.714102029800415, "reward_meter_mean": 0.9980402588844299, "reward_meter_std": 0.00200492306612432, "reward_count_adherence_mean": 0.7857142686843872, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.910714328289032, "reward_repeat_penalty_std": 0.06458108127117157, "reward_total_composite_mean": 0.714102029800415, "reward_total_composite_std": 0.04969966039061546} {"timestamp_utc": "2026-04-12T02:08:59Z", "mode": "train", "global_step": 2478, "epoch": 0.09953006386311604, "loss": 0.0048, "grad_norm": 6.063255786895752, "learning_rate": 2.4939393939393943e-06, "num_tokens": 5615759.0, "completions/mean_length": 120.0, "completions/min_length": 117.0, "completions/max_length": 123.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.0, "completions/min_terminated_length": 117.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.9968112111091614, "rewards/meter/std": 0.004492159932851791, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9968112111091614, "rewards/total_composite/std": 0.004492159932851791, "reward": 0.9968112111091614, "reward_std": 0.004492142703384161, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03631357103586197, "sampling/sampling_logp_difference/max": 0.8470320701599121, "sampling/importance_sampling_ratio/min": 0.42868536710739136, "sampling/importance_sampling_ratio/mean": 1.0102128982543945, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3629257455468178, "clip_ratio/low_mean": 0.0020833334419876337, "clip_ratio/low_min": 0.0020833334419876337, "clip_ratio/high_mean": 0.036573544377461076, "clip_ratio/high_max": 0.036573544377461076, "clip_ratio/region_mean": 0.03865687781944871, "reward_total_mean": 0.9968112111091614, "reward_meter_mean": 0.9968112111091614, "reward_meter_std": 0.004492159932851791, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9968112111091614, "reward_total_composite_std": 0.004492159932851791} {"timestamp_utc": "2026-04-12T02:09:04Z", "mode": "train", "global_step": 2479, "epoch": 0.09957022934490099, "loss": 0.0172, "grad_norm": 5.862101078033447, "learning_rate": 2.490909090909091e-06, "num_tokens": 5617602.0, "completions/mean_length": 80.375, "completions/min_length": 78.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.375, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.9985413551330566, "rewards/meter/std": 0.0006061770836822689, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9985413551330566, "rewards/total_composite/std": 0.0006061770836822689, "reward": 0.9985413551330566, "reward_std": 0.0006061761523596942, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03767424076795578, "sampling/sampling_logp_difference/max": 1.3134748935699463, "sampling/importance_sampling_ratio/min": 0.2688840925693512, "sampling/importance_sampling_ratio/mean": 1.0086257457733154, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.26199326291680336, "clip_ratio/low_mean": 0.010923440335318446, "clip_ratio/low_min": 0.010923440335318446, "clip_ratio/high_mean": 0.00933689041994512, "clip_ratio/high_max": 0.00933689041994512, "clip_ratio/region_mean": 0.020260330755263567, "reward_total_mean": 0.9985413551330566, "reward_meter_mean": 0.9985413551330566, "reward_meter_std": 0.0006061770836822689, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9985413551330566, "reward_total_composite_std": 0.0006061770836822689} {"timestamp_utc": "2026-04-12T02:09:09Z", "mode": "train", "global_step": 2480, "epoch": 0.09961039482668595, "loss": 0.01, "grad_norm": 2.8270602226257324, "learning_rate": 2.487878787878788e-06, "num_tokens": 5619995.0, "completions/mean_length": 127.125, "completions/min_length": 126.0, "completions/max_length": 130.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.125, "completions/min_terminated_length": 126.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.7682651281356812, "rewards/meter/std": 0.1492764949798584, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.7446907758712769, "rewards/total_composite/std": 0.17325718700885773, "reward": 0.7446907758712769, "reward_std": 0.17325717210769653, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03481648862361908, "sampling/sampling_logp_difference/max": 1.9265758991241455, "sampling/importance_sampling_ratio/min": 0.14564606547355652, "sampling/importance_sampling_ratio/mean": 1.0069787502288818, "sampling/importance_sampling_ratio/max": 1.9678964614868164, "entropy": 0.25193316861987114, "clip_ratio/low_mean": 0.009811883908696473, "clip_ratio/low_min": 0.009811883908696473, "clip_ratio/high_mean": 0.016681208508089185, "clip_ratio/high_max": 0.016681208508089185, "clip_ratio/region_mean": 0.026493092416785657, "reward_total_mean": 0.7446907758712769, "reward_meter_mean": 0.7682651281356812, "reward_meter_std": 0.1492764949798584, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.7446907758712769, "reward_total_composite_std": 0.17325718700885773} {"timestamp_utc": "2026-04-12T02:09:13Z", "mode": "train", "global_step": 2481, "epoch": 0.0996505603084709, "loss": 0.0178, "grad_norm": 5.6794843673706055, "learning_rate": 2.4848484848484848e-06, "num_tokens": 5621792.0, "completions/mean_length": 65.625, "completions/min_length": 65.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.625, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9421236515045166, "rewards/meter/std": 0.09346310049295425, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9421236515045166, "rewards/total_composite/std": 0.09346310049295425, "reward": 0.9421236515045166, "reward_std": 0.09346310794353485, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02599693275988102, "sampling/sampling_logp_difference/max": 0.5891776084899902, "sampling/importance_sampling_ratio/min": 0.5547833442687988, "sampling/importance_sampling_ratio/mean": 1.0088194608688354, "sampling/importance_sampling_ratio/max": 1.7557963132858276, "entropy": 0.1946810930967331, "clip_ratio/low_mean": 0.009328358108177781, "clip_ratio/low_min": 0.009328358108177781, "clip_ratio/high_mean": 0.02106643421575427, "clip_ratio/high_max": 0.02106643421575427, "clip_ratio/region_mean": 0.03039479232393205, "reward_total_mean": 0.9421236515045166, "reward_meter_mean": 0.9421236515045166, "reward_meter_std": 0.09346310049295425, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9421236515045166, "reward_total_composite_std": 0.09346310049295425} {"timestamp_utc": "2026-04-12T02:09:18Z", "mode": "train", "global_step": 2482, "epoch": 0.09969072579025585, "loss": 0.1809, "grad_norm": 28.66371726989746, "learning_rate": 2.481818181818182e-06, "num_tokens": 5623580.0, "completions/mean_length": 60.5, "completions/min_length": 54.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.5, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.7704848051071167, "rewards/meter/std": 0.03176296874880791, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6806069612503052, "rewards/total_composite/std": 0.19818444550037384, "reward": 0.6806069612503052, "reward_std": 0.19818443059921265, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.001127883093431592, "sampling/sampling_logp_difference/max": 0.07983016967773438, "sampling/importance_sampling_ratio/min": 0.9490934610366821, "sampling/importance_sampling_ratio/mean": 1.0007379055023193, "sampling/importance_sampling_ratio/max": 1.083103060722351, "entropy": 0.007361570082139224, "clip_ratio/low_mean": 0.0015625000232830644, "clip_ratio/low_min": 0.0015625000232830644, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0015625000232830644, "reward_total_mean": 0.6806069612503052, "reward_meter_mean": 0.7704848051071167, "reward_meter_std": 0.03176296874880791, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6806069612503052, "reward_total_composite_std": 0.19818444550037384} {"timestamp_utc": "2026-04-12T02:09:23Z", "mode": "train", "global_step": 2483, "epoch": 0.09973089127204081, "loss": 0.0225, "grad_norm": 3.7192165851593018, "learning_rate": 2.478787878787879e-06, "num_tokens": 5625374.0, "completions/mean_length": 61.25, "completions/min_length": 56.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.25, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9937635660171509, "rewards/meter/std": 0.0006408250774256885, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9937635660171509, "rewards/total_composite/std": 0.0006408250774256885, "reward": 0.9937635660171509, "reward_std": 0.0006408333429135382, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015063888393342495, "sampling/sampling_logp_difference/max": 1.5699043273925781, "sampling/importance_sampling_ratio/min": 0.20806507766246796, "sampling/importance_sampling_ratio/mean": 0.998705267906189, "sampling/importance_sampling_ratio/max": 1.8052740097045898, "entropy": 0.049932122230529785, "clip_ratio/low_mean": 0.004032257944345474, "clip_ratio/low_min": 0.004032257944345474, "clip_ratio/high_mean": 0.00828052987344563, "clip_ratio/high_max": 0.00828052987344563, "clip_ratio/region_mean": 0.012312787817791104, "reward_total_mean": 0.9937635660171509, "reward_meter_mean": 0.9937635660171509, "reward_meter_std": 0.0006408250774256885, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9937635660171509, "reward_total_composite_std": 0.0006408250774256885} {"timestamp_utc": "2026-04-12T02:09:27Z", "mode": "train", "global_step": 2484, "epoch": 0.09977105675382576, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.475757575757576e-06, "num_tokens": 5627126.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00020638658315874636, "sampling/sampling_logp_difference/max": 0.007955902256071568, "sampling/importance_sampling_ratio/min": 0.9975246787071228, "sampling/importance_sampling_ratio/mean": 1.0001825094223022, "sampling/importance_sampling_ratio/max": 1.00798761844635, "entropy": 0.002001831730012782, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:09:36Z", "mode": "train", "global_step": 2485, "epoch": 0.09981122223561072, "loss": -0.0278, "grad_norm": 1.8019484281539917, "learning_rate": 2.472727272727273e-06, "num_tokens": 5631659.0, "completions/mean_length": 363.625, "completions/min_length": 343.0, "completions/max_length": 393.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 363.625, "completions/min_terminated_length": 343.0, "completions/max_terminated_length": 393.0, "rewards/meter/mean": 0.9972596764564514, "rewards/meter/std": 0.004136989358812571, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.04629101976752281, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9709967374801636, "rewards/repeat_penalty/std": 0.031024247407913208, "rewards/total_composite/mean": 0.8964345455169678, "rewards/total_composite/std": 0.0664711743593216, "reward": 0.8964345455169678, "reward_std": 0.0664711743593216, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.052424363791942596, "sampling/sampling_logp_difference/max": 2.488617181777954, "sampling/importance_sampling_ratio/min": 0.08302469551563263, "sampling/importance_sampling_ratio/mean": 1.0112888813018799, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4474870078265667, "clip_ratio/low_mean": 0.01650261995382607, "clip_ratio/low_min": 0.01650261995382607, "clip_ratio/high_mean": 0.021971354726701975, "clip_ratio/high_max": 0.021971354726701975, "clip_ratio/region_mean": 0.038473974680528045, "reward_total_mean": 0.8964345455169678, "reward_meter_mean": 0.9972596764564514, "reward_meter_std": 0.004136989358812571, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.04629101976752281, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9709967374801636, "reward_repeat_penalty_std": 0.031024247407913208, "reward_total_composite_mean": 0.8964345455169678, "reward_total_composite_std": 0.0664711743593216} {"timestamp_utc": "2026-04-12T02:09:41Z", "mode": "train", "global_step": 2486, "epoch": 0.09985138771739567, "loss": 0.0133, "grad_norm": 2.246661901473999, "learning_rate": 2.46969696969697e-06, "num_tokens": 5634271.0, "completions/mean_length": 158.5, "completions/min_length": 156.0, "completions/max_length": 161.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 158.5, "completions/min_terminated_length": 156.0, "completions/max_terminated_length": 161.0, "rewards/meter/mean": 0.9984291791915894, "rewards/meter/std": 0.0007291806978173554, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9984291791915894, "rewards/total_composite/std": 0.0007291806978173554, "reward": 0.9984291791915894, "reward_std": 0.0007291696383617818, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.049818817526102066, "sampling/sampling_logp_difference/max": 1.498807430267334, "sampling/importance_sampling_ratio/min": 0.223396435379982, "sampling/importance_sampling_ratio/mean": 1.0025769472122192, "sampling/importance_sampling_ratio/max": 1.7283272743225098, "entropy": 0.45672084018588066, "clip_ratio/low_mean": 0.005464080721139908, "clip_ratio/low_min": 0.005464080721139908, "clip_ratio/high_mean": 0.03322292352095246, "clip_ratio/high_max": 0.03322292352095246, "clip_ratio/region_mean": 0.03868700424209237, "reward_total_mean": 0.9984291791915894, "reward_meter_mean": 0.9984291791915894, "reward_meter_std": 0.0007291806978173554, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9984291791915894, "reward_total_composite_std": 0.0007291806978173554} {"timestamp_utc": "2026-04-12T02:09:46Z", "mode": "train", "global_step": 2487, "epoch": 0.09989155319918062, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.466666666666667e-06, "num_tokens": 5636167.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00024826714070513844, "sampling/sampling_logp_difference/max": 0.011286328546702862, "sampling/importance_sampling_ratio/min": 0.9940236806869507, "sampling/importance_sampling_ratio/mean": 1.0001945495605469, "sampling/importance_sampling_ratio/max": 1.0113502740859985, "entropy": 0.0025851023383438587, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:09:51Z", "mode": "train", "global_step": 2488, "epoch": 0.09993171868096558, "loss": -0.0124, "grad_norm": 0.9873250126838684, "learning_rate": 2.463636363636364e-06, "num_tokens": 5638299.0, "completions/mean_length": 97.5, "completions/min_length": 94.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.5, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9964094758033752, "rewards/meter/std": 0.004482356831431389, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9964094758033752, "rewards/total_composite/std": 0.004482356831431389, "reward": 0.9964094758033752, "reward_std": 0.00448233587667346, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0035261192824691534, "sampling/sampling_logp_difference/max": 1.280348777770996, "sampling/importance_sampling_ratio/min": 0.2779403328895569, "sampling/importance_sampling_ratio/mean": 0.9996676445007324, "sampling/importance_sampling_ratio/max": 1.091353416442871, "entropy": 0.01773504843004048, "clip_ratio/low_mean": 0.0013297871919348836, "clip_ratio/low_min": 0.0013297871919348836, "clip_ratio/high_mean": 0.0038265305338427424, "clip_ratio/high_max": 0.0038265305338427424, "clip_ratio/region_mean": 0.005156317725777626, "reward_total_mean": 0.9964094758033752, "reward_meter_mean": 0.9964094758033752, "reward_meter_std": 0.004482356831431389, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9964094758033752, "reward_total_composite_std": 0.004482356831431389} {"timestamp_utc": "2026-04-12T02:09:55Z", "mode": "train", "global_step": 2489, "epoch": 0.09997188416275053, "loss": 0.0144, "grad_norm": 19.695940017700195, "learning_rate": 2.4606060606060607e-06, "num_tokens": 5639817.0, "completions/mean_length": 23.75, "completions/min_length": 21.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.75, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9360402822494507, "rewards/meter/std": 0.019930923357605934, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9360402822494507, "rewards/total_composite/std": 0.019930923357605934, "reward": 0.9360402822494507, "reward_std": 0.019930915907025337, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09426023066043854, "sampling/sampling_logp_difference/max": 1.1065075397491455, "sampling/importance_sampling_ratio/min": 0.3307119607925415, "sampling/importance_sampling_ratio/mean": 1.0085498094558716, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5839281938970089, "clip_ratio/low_mean": 0.05386740108951926, "clip_ratio/low_min": 0.05386740108951926, "clip_ratio/high_mean": 0.03615045174956322, "clip_ratio/high_max": 0.03615045174956322, "clip_ratio/region_mean": 0.09001785283908248, "reward_total_mean": 0.9360402822494507, "reward_meter_mean": 0.9360402822494507, "reward_meter_std": 0.019930923357605934, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9360402822494507, "reward_total_composite_std": 0.019930923357605934} {"timestamp_utc": "2026-04-12T02:10:00Z", "mode": "train", "global_step": 2490, "epoch": 0.10001204964453549, "loss": 0.0052, "grad_norm": 1.8480373620986938, "learning_rate": 2.457575757575758e-06, "num_tokens": 5642037.0, "completions/mean_length": 107.5, "completions/min_length": 106.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.5, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.9976269602775574, "rewards/meter/std": 0.0004338659346103668, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9727107882499695, "rewards/total_composite/std": 0.0708698034286499, "reward": 0.9727107882499695, "reward_std": 0.0708698257803917, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01844092458486557, "sampling/sampling_logp_difference/max": 2.3559987545013428, "sampling/importance_sampling_ratio/min": 0.09479877352714539, "sampling/importance_sampling_ratio/mean": 1.003526210784912, "sampling/importance_sampling_ratio/max": 1.686613917350769, "entropy": 0.12307433690875769, "clip_ratio/low_mean": 0.0022935778833925724, "clip_ratio/low_min": 0.0022935778833925724, "clip_ratio/high_mean": 0.008156140334904194, "clip_ratio/high_max": 0.008156140334904194, "clip_ratio/region_mean": 0.010449718218296766, "reward_total_mean": 0.9727107882499695, "reward_meter_mean": 0.9976269602775574, "reward_meter_std": 0.0004338659346103668, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9727107882499695, "reward_total_composite_std": 0.0708698034286499} {"timestamp_utc": "2026-04-12T02:10:05Z", "mode": "train", "global_step": 2491, "epoch": 0.10005221512632044, "loss": -0.0008, "grad_norm": 2.2775914669036865, "learning_rate": 2.454545454545455e-06, "num_tokens": 5643821.0, "completions/mean_length": 62.0, "completions/min_length": 62.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9934873580932617, "rewards/meter/std": 0.0007373938569799066, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9934873580932617, "rewards/total_composite/std": 0.0007373938569799066, "reward": 0.9934873580932617, "reward_std": 0.000737398921046406, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007004985585808754, "sampling/sampling_logp_difference/max": 0.4697551727294922, "sampling/importance_sampling_ratio/min": 0.6251553297042847, "sampling/importance_sampling_ratio/mean": 1.001605749130249, "sampling/importance_sampling_ratio/max": 1.28243088722229, "entropy": 0.040033137891441584, "clip_ratio/low_mean": 0.004032257944345474, "clip_ratio/low_min": 0.004032257944345474, "clip_ratio/high_mean": 0.002016128972172737, "clip_ratio/high_max": 0.002016128972172737, "clip_ratio/region_mean": 0.006048386916518211, "reward_total_mean": 0.9934873580932617, "reward_meter_mean": 0.9934873580932617, "reward_meter_std": 0.0007373938569799066, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9934873580932617, "reward_total_composite_std": 0.0007373938569799066} {"timestamp_utc": "2026-04-12T02:10:10Z", "mode": "train", "global_step": 2492, "epoch": 0.1000923806081054, "loss": -0.006, "grad_norm": 3.0206403732299805, "learning_rate": 2.4515151515151516e-06, "num_tokens": 5646242.0, "completions/mean_length": 125.625, "completions/min_length": 123.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 125.625, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.9930568337440491, "rewards/meter/std": 0.00019196225912310183, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.9575905799865723, "rewards/total_composite/std": 0.06567274034023285, "reward": 0.9575905799865723, "reward_std": 0.06567277014255524, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018952438607811928, "sampling/sampling_logp_difference/max": 2.4076790809631348, "sampling/importance_sampling_ratio/min": 0.09002399444580078, "sampling/importance_sampling_ratio/mean": 1.001326322555542, "sampling/importance_sampling_ratio/max": 1.629332184791565, "entropy": 0.10137755703181028, "clip_ratio/low_mean": 0.0010080644860863686, "clip_ratio/low_min": 0.0010080644860863686, "clip_ratio/high_mean": 0.01584376988466829, "clip_ratio/high_max": 0.01584376988466829, "clip_ratio/region_mean": 0.01685183437075466, "reward_total_mean": 0.9575905799865723, "reward_meter_mean": 0.9930568337440491, "reward_meter_std": 0.00019196225912310183, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.9575905799865723, "reward_total_composite_std": 0.06567274034023285} {"timestamp_utc": "2026-04-12T02:10:16Z", "mode": "train", "global_step": 2493, "epoch": 0.10013254608989035, "loss": 0.0227, "grad_norm": 3.180861234664917, "learning_rate": 2.4484848484848485e-06, "num_tokens": 5649068.0, "completions/mean_length": 162.25, "completions/min_length": 154.0, "completions/max_length": 172.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 162.25, "completions/min_terminated_length": 154.0, "completions/max_terminated_length": 172.0, "rewards/meter/mean": 0.904288649559021, "rewards/meter/std": 0.09930223226547241, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.881544828414917, "rewards/total_composite/std": 0.1262788623571396, "reward": 0.881544828414917, "reward_std": 0.1262788623571396, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.041033390909433365, "sampling/sampling_logp_difference/max": 1.6256977319717407, "sampling/importance_sampling_ratio/min": 0.19677433371543884, "sampling/importance_sampling_ratio/mean": 1.0083729028701782, "sampling/importance_sampling_ratio/max": 1.7914950847625732, "entropy": 0.4163212776184082, "clip_ratio/low_mean": 0.005151562392711639, "clip_ratio/low_min": 0.005151562392711639, "clip_ratio/high_mean": 0.025025799637660384, "clip_ratio/high_max": 0.025025799637660384, "clip_ratio/region_mean": 0.030177362030372024, "reward_total_mean": 0.881544828414917, "reward_meter_mean": 0.904288649559021, "reward_meter_std": 0.09930223226547241, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_total_composite_mean": 0.881544828414917, "reward_total_composite_std": 0.1262788623571396} {"timestamp_utc": "2026-04-12T02:10:20Z", "mode": "train", "global_step": 2494, "epoch": 0.1001727115716753, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.4454545454545457e-06, "num_tokens": 5650820.0, "completions/mean_length": 67.0, "completions/min_length": 67.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981522560119629, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981522560119629, "rewards/total_composite/std": 0.0, "reward": 0.9981522560119629, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0011890914756804705, "sampling/sampling_logp_difference/max": 0.09505826234817505, "sampling/importance_sampling_ratio/min": 0.9093199372291565, "sampling/importance_sampling_ratio/mean": 1.000666856765747, "sampling/importance_sampling_ratio/max": 1.0194705724716187, "entropy": 0.010702198022045195, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9981522560119629, "reward_meter_mean": 0.9981522560119629, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981522560119629, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:10:25Z", "mode": "train", "global_step": 2495, "epoch": 0.10021287705346026, "loss": 0.0059, "grad_norm": 1.8723995685577393, "learning_rate": 2.4424242424242426e-06, "num_tokens": 5652882.0, "completions/mean_length": 94.75, "completions/min_length": 93.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.75, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.9930741786956787, "rewards/meter/std": 0.0006547096418216825, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9682368040084839, "rewards/total_composite/std": 0.07008406519889832, "reward": 0.9682368040084839, "reward_std": 0.07008406519889832, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018359439447522163, "sampling/sampling_logp_difference/max": 1.3887449502944946, "sampling/importance_sampling_ratio/min": 0.24938809871673584, "sampling/importance_sampling_ratio/mean": 1.0004006624221802, "sampling/importance_sampling_ratio/max": 1.5056626796722412, "entropy": 0.10162492096424103, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.014545450918376446, "clip_ratio/high_max": 0.014545450918376446, "clip_ratio/region_mean": 0.014545450918376446, "reward_total_mean": 0.9682368040084839, "reward_meter_mean": 0.9930741786956787, "reward_meter_std": 0.0006547096418216825, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9682368040084839, "reward_total_composite_std": 0.07008406519889832} {"timestamp_utc": "2026-04-12T02:10:30Z", "mode": "train", "global_step": 2496, "epoch": 0.10025304253524521, "loss": 0.0159, "grad_norm": 3.207080602645874, "learning_rate": 2.4393939393939394e-06, "num_tokens": 5655174.0, "completions/mean_length": 105.5, "completions/min_length": 98.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.5, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.9972423315048218, "rewards/meter/std": 0.0021226475946605206, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.9473263025283813, "rewards/total_composite/std": 0.09178309142589569, "reward": 0.9473263025283813, "reward_std": 0.0917830839753151, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03410215303301811, "sampling/sampling_logp_difference/max": 1.372025489807129, "sampling/importance_sampling_ratio/min": 0.2535927891731262, "sampling/importance_sampling_ratio/mean": 1.0094503164291382, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23731265030801296, "clip_ratio/low_mean": 0.004629629664123058, "clip_ratio/low_min": 0.004629629664123058, "clip_ratio/high_mean": 0.02076363516971469, "clip_ratio/high_max": 0.02076363516971469, "clip_ratio/region_mean": 0.025393264833837748, "reward_total_mean": 0.9473263025283813, "reward_meter_mean": 0.9972423315048218, "reward_meter_std": 0.0021226475946605206, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.9473263025283813, "reward_total_composite_std": 0.09178309142589569} {"timestamp_utc": "2026-04-12T02:10:36Z", "mode": "train", "global_step": 2497, "epoch": 0.10029320801703016, "loss": 0.0681, "grad_norm": 5.552990913391113, "learning_rate": 2.4363636363636366e-06, "num_tokens": 5658127.0, "completions/mean_length": 175.125, "completions/min_length": 166.0, "completions/max_length": 206.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 175.125, "completions/min_terminated_length": 166.0, "completions/max_terminated_length": 206.0, "rewards/meter/mean": 0.9639856219291687, "rewards/meter/std": 0.09431485831737518, "rewards/count_adherence/mean": 0.8541666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8185431361198425, "rewards/total_composite/std": 0.03557335212826729, "reward": 0.8185431361198425, "reward_std": 0.035573359578847885, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0666884258389473, "sampling/sampling_logp_difference/max": 2.1122121810913086, "sampling/importance_sampling_ratio/min": 0.12097006291151047, "sampling/importance_sampling_ratio/mean": 1.0070842504501343, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5811385102570057, "clip_ratio/low_mean": 0.004247572738677263, "clip_ratio/low_min": 0.004247572738677263, "clip_ratio/high_mean": 0.05119223310612142, "clip_ratio/high_max": 0.05119223310612142, "clip_ratio/region_mean": 0.055439805844798684, "reward_total_mean": 0.8185431361198425, "reward_meter_mean": 0.9639856219291687, "reward_meter_std": 0.09431485831737518, "reward_count_adherence_mean": 0.8541666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8185431361198425, "reward_total_composite_std": 0.03557335212826729} {"timestamp_utc": "2026-04-12T02:10:41Z", "mode": "train", "global_step": 2498, "epoch": 0.10033337349881512, "loss": -0.0013, "grad_norm": 3.1228280067443848, "learning_rate": 2.4333333333333335e-06, "num_tokens": 5660216.0, "completions/mean_length": 97.125, "completions/min_length": 95.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.125, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.991081714630127, "rewards/meter/std": 0.004786557983607054, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.9413519501686096, "rewards/total_composite/std": 0.09001474827528, "reward": 0.9413519501686096, "reward_std": 0.09001474827528, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031947266310453415, "sampling/sampling_logp_difference/max": 0.8252577781677246, "sampling/importance_sampling_ratio/min": 0.43812206387519836, "sampling/importance_sampling_ratio/mean": 1.0101373195648193, "sampling/importance_sampling_ratio/max": 1.695669412612915, "entropy": 0.30781341157853603, "clip_ratio/low_mean": 0.0013020833721384406, "clip_ratio/low_min": 0.0013020833721384406, "clip_ratio/high_mean": 0.01925921393558383, "clip_ratio/high_max": 0.01925921393558383, "clip_ratio/region_mean": 0.02056129730772227, "reward_total_mean": 0.9413519501686096, "reward_meter_mean": 0.991081714630127, "reward_meter_std": 0.004786557983607054, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.9413519501686096, "reward_total_composite_std": 0.09001474827528} {"timestamp_utc": "2026-04-12T02:10:46Z", "mode": "train", "global_step": 2499, "epoch": 0.10037353898060007, "loss": 0.008, "grad_norm": 3.2906627655029297, "learning_rate": 2.4303030303030307e-06, "num_tokens": 5661956.0, "completions/mean_length": 67.5, "completions/min_length": 67.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.5, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9980784058570862, "rewards/meter/std": 0.00020892193424515426, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980784058570862, "rewards/total_composite/std": 0.00020892193424515426, "reward": 0.9980784058570862, "reward_std": 0.00020891590975224972, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0033059667330235243, "sampling/sampling_logp_difference/max": 1.0325679779052734, "sampling/importance_sampling_ratio/min": 0.35609135031700134, "sampling/importance_sampling_ratio/mean": 0.9999598860740662, "sampling/importance_sampling_ratio/max": 1.0832960605621338, "entropy": 0.015344980405643582, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9980784058570862, "reward_meter_mean": 0.9980784058570862, "reward_meter_std": 0.00020892193424515426, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9980784058570862, "reward_total_composite_std": 0.00020892193424515426} {"timestamp_utc": "2026-04-12T02:10:51Z", "mode": "train", "global_step": 2500, "epoch": 0.10041370446238503, "loss": -0.0048, "grad_norm": 3.226564884185791, "learning_rate": 2.4272727272727276e-06, "num_tokens": 5664327.0, "completions/mean_length": 131.375, "completions/min_length": 130.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.375, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.998822808265686, "rewards/meter/std": 0.0007065999088808894, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.910714328289032, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9096584916114807, "rewards/total_composite/std": 0.07408842444419861, "reward": 0.9096584916114807, "reward_std": 0.07408840954303741, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01683160476386547, "sampling/sampling_logp_difference/max": 1.742103099822998, "sampling/importance_sampling_ratio/min": 0.17515164613723755, "sampling/importance_sampling_ratio/mean": 0.9981679916381836, "sampling/importance_sampling_ratio/max": 1.8086941242218018, "entropy": 0.05698791332542896, "clip_ratio/low_mean": 0.011479852895718068, "clip_ratio/low_min": 0.011479852895718068, "clip_ratio/high_mean": 0.004727728548459709, "clip_ratio/high_max": 0.004727728548459709, "clip_ratio/region_mean": 0.016207581444177777, "reward_total_mean": 0.9096584916114807, "reward_meter_mean": 0.998822808265686, "reward_meter_std": 0.0007065999088808894, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.910714328289032, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.9096584916114807, "reward_total_composite_std": 0.07408842444419861} {"timestamp_utc": "2026-04-12T02:12:07Z", "mode": "eval", "global_step": 2500, "epoch": 0.10041370446238503, "eval_loss": NaN, "eval_runtime": 75.4547, "eval_samples_per_second": 1.378, "eval_steps_per_second": 0.172, "eval_num_tokens": 5664327.0, "eval_completions/mean_length": 210.92307692307693, "eval_completions/min_length": 62.38461538461539, "eval_completions/max_length": 405.3076923076923, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 210.92307692307693, "eval_completions/min_terminated_length": 62.38461538461539, "eval_completions/max_terminated_length": 405.3076923076923, "eval_rewards/meter/mean": 0.7411972742814285, "eval_rewards/meter/std": 0.36808492687459177, "eval_rewards/count_adherence/mean": 0.9494116031206571, "eval_rewards/count_adherence/std": 0.07668151821081455, "eval_rewards/arabic_clean/mean": 0.9711538461538461, "eval_rewards/arabic_clean/std": 0.08158924258672275, "eval_rewards/repeat_penalty/mean": 0.9134463484470661, "eval_rewards/repeat_penalty/std": 0.10245848590364823, "eval_rewards/total_composite/mean": 0.6316862404346466, "eval_rewards/total_composite/std": 0.3585601345850871, "eval_reward": 0.6316862404346466, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.03223242281148067, "eval_sampling/sampling_logp_difference/max": 1.1953271352327788, "eval_sampling/importance_sampling_ratio/min": 0.31562885412803066, "eval_sampling/importance_sampling_ratio/mean": 1.0096348799191988, "eval_sampling/importance_sampling_ratio/max": 1.5564926862716675, "eval_entropy": 0.37157516181468964, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6316862404346466, "eval_reward_meter_mean": 0.7411972742814285, "eval_reward_meter_std": 0.36808492687459177, "eval_reward_count_adherence_mean": 0.9494116031206571, "eval_reward_count_adherence_std": 0.07668151821081455, "eval_reward_arabic_clean_mean": 0.9711538461538461, "eval_reward_arabic_clean_std": 0.08158924258672275, "eval_reward_repeat_penalty_mean": 0.9134463484470661, "eval_reward_repeat_penalty_std": 0.10245848590364823, "eval_reward_total_composite_mean": 0.6316862404346466, "eval_reward_total_composite_std": 0.3585601345850871} {"timestamp_utc": "2026-04-12T02:12:13Z", "mode": "train", "global_step": 2501, "epoch": 0.10045386994416998, "loss": 0.018, "grad_norm": 4.538857460021973, "learning_rate": 2.4242424242424244e-06, "num_tokens": 5666190.0, "completions/mean_length": 66.875, "completions/min_length": 64.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.875, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9774847626686096, "rewards/meter/std": 0.03591068461537361, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9774847626686096, "rewards/total_composite/std": 0.03591068461537361, "reward": 0.9774847626686096, "reward_std": 0.03591068089008331, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04367325082421303, "sampling/sampling_logp_difference/max": 0.8173356056213379, "sampling/importance_sampling_ratio/min": 0.44160670042037964, "sampling/importance_sampling_ratio/mean": 1.0134010314941406, "sampling/importance_sampling_ratio/max": 1.6371313333511353, "entropy": 0.3523239102214575, "clip_ratio/low_mean": 0.003677265834994614, "clip_ratio/low_min": 0.003677265834994614, "clip_ratio/high_mean": 0.02857714961282909, "clip_ratio/high_max": 0.02857714961282909, "clip_ratio/region_mean": 0.0322544154478237, "reward_total_mean": 0.9774847626686096, "reward_meter_mean": 0.9774847626686096, "reward_meter_std": 0.03591068461537361, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9774847626686096, "reward_total_composite_std": 0.03591068461537361} {"timestamp_utc": "2026-04-12T02:12:18Z", "mode": "train", "global_step": 2502, "epoch": 0.10049403542595493, "loss": 0.0026, "grad_norm": 4.3011322021484375, "learning_rate": 2.4212121212121216e-06, "num_tokens": 5668128.0, "completions/mean_length": 67.25, "completions/min_length": 65.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.25, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9222630262374878, "rewards/meter/std": 0.14857283234596252, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9222630262374878, "rewards/total_composite/std": 0.14857283234596252, "reward": 0.9222630262374878, "reward_std": 0.14857284724712372, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04170745983719826, "sampling/sampling_logp_difference/max": 0.8916559219360352, "sampling/importance_sampling_ratio/min": 0.4099763035774231, "sampling/importance_sampling_ratio/mean": 1.0067012310028076, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3921285644173622, "clip_ratio/low_mean": 0.013037330703809857, "clip_ratio/low_min": 0.013037330703809857, "clip_ratio/high_mean": 0.029540141811594367, "clip_ratio/high_max": 0.029540141811594367, "clip_ratio/region_mean": 0.042577472515404224, "reward_total_mean": 0.9222630262374878, "reward_meter_mean": 0.9222630262374878, "reward_meter_std": 0.14857283234596252, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9222630262374878, "reward_total_composite_std": 0.14857283234596252} {"timestamp_utc": "2026-04-12T02:12:23Z", "mode": "train", "global_step": 2503, "epoch": 0.10053420090773989, "loss": 0.0043, "grad_norm": 4.316682815551758, "learning_rate": 2.4181818181818185e-06, "num_tokens": 5669907.0, "completions/mean_length": 71.375, "completions/min_length": 70.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.375, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9978486895561218, "rewards/meter/std": 0.00048536990652792156, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978486895561218, "rewards/total_composite/std": 0.00048536990652792156, "reward": 0.9978486895561218, "reward_std": 0.00048536164104007185, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021549848839640617, "sampling/sampling_logp_difference/max": 0.9752523899078369, "sampling/importance_sampling_ratio/min": 0.37709715962409973, "sampling/importance_sampling_ratio/mean": 1.0028010606765747, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.14498437847942114, "clip_ratio/low_mean": 0.008732197224162519, "clip_ratio/low_min": 0.008732197224162519, "clip_ratio/high_mean": 0.0035211266949772835, "clip_ratio/high_max": 0.0035211266949772835, "clip_ratio/region_mean": 0.012253323919139802, "reward_total_mean": 0.9978486895561218, "reward_meter_mean": 0.9978486895561218, "reward_meter_std": 0.00048536990652792156, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9978486895561218, "reward_total_composite_std": 0.00048536990652792156} {"timestamp_utc": "2026-04-12T02:12:27Z", "mode": "train", "global_step": 2504, "epoch": 0.10057436638952484, "loss": -0.0032, "grad_norm": 6.806858539581299, "learning_rate": 2.4151515151515153e-06, "num_tokens": 5671433.0, "completions/mean_length": 36.75, "completions/min_length": 36.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.75, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.9697108268737793, "rewards/meter/std": 0.01697305217385292, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9697108268737793, "rewards/total_composite/std": 0.01697305217385292, "reward": 0.9697108268737793, "reward_std": 0.01697305217385292, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03803471848368645, "sampling/sampling_logp_difference/max": 0.5422968864440918, "sampling/importance_sampling_ratio/min": 0.5814112424850464, "sampling/importance_sampling_ratio/mean": 1.0082935094833374, "sampling/importance_sampling_ratio/max": 1.5999165773391724, "entropy": 0.32645551674067974, "clip_ratio/low_mean": 0.02393018058501184, "clip_ratio/low_min": 0.02393018058501184, "clip_ratio/high_mean": 0.02037855191156268, "clip_ratio/high_max": 0.02037855191156268, "clip_ratio/region_mean": 0.04430873249657452, "reward_total_mean": 0.9697108268737793, "reward_meter_mean": 0.9697108268737793, "reward_meter_std": 0.01697305217385292, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9697108268737793, "reward_total_composite_std": 0.01697305217385292} {"timestamp_utc": "2026-04-12T02:12:32Z", "mode": "train", "global_step": 2505, "epoch": 0.1006145318713098, "loss": -0.0003, "grad_norm": 0.20346412062644958, "learning_rate": 2.412121212121212e-06, "num_tokens": 5673402.0, "completions/mean_length": 67.125, "completions/min_length": 67.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.125, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9981461763381958, "rewards/meter/std": 1.717484337859787e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981461763381958, "rewards/total_composite/std": 1.717484337859787e-05, "reward": 0.9981461763381958, "reward_std": 1.7183874660986476e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00340123544447124, "sampling/sampling_logp_difference/max": 0.710139274597168, "sampling/importance_sampling_ratio/min": 0.4915757179260254, "sampling/importance_sampling_ratio/mean": 0.9994654655456543, "sampling/importance_sampling_ratio/max": 1.091694712638855, "entropy": 0.013536597951315343, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0018382353009656072, "clip_ratio/high_max": 0.0018382353009656072, "clip_ratio/region_mean": 0.0018382353009656072, "reward_total_mean": 0.9981461763381958, "reward_meter_mean": 0.9981461763381958, "reward_meter_std": 1.717484337859787e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981461763381958, "reward_total_composite_std": 1.717484337859787e-05} {"timestamp_utc": "2026-04-12T02:12:37Z", "mode": "train", "global_step": 2506, "epoch": 0.10065469735309475, "loss": -0.0096, "grad_norm": 6.047022342681885, "learning_rate": 2.4090909090909094e-06, "num_tokens": 5675142.0, "completions/mean_length": 67.5, "completions/min_length": 65.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.5, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9976764917373657, "rewards/meter/std": 0.004249152261763811, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9976764917373657, "rewards/total_composite/std": 0.004249152261763811, "reward": 0.9976764917373657, "reward_std": 0.004249148536473513, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04069520905613899, "sampling/sampling_logp_difference/max": 2.3988044261932373, "sampling/importance_sampling_ratio/min": 0.09082647413015366, "sampling/importance_sampling_ratio/mean": 1.0117424726486206, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3341045305132866, "clip_ratio/low_mean": 0.003846153849735856, "clip_ratio/low_min": 0.003846153849735856, "clip_ratio/high_mean": 0.02579016692470759, "clip_ratio/high_max": 0.02579016692470759, "clip_ratio/region_mean": 0.029636320774443448, "reward_total_mean": 0.9976764917373657, "reward_meter_mean": 0.9976764917373657, "reward_meter_std": 0.004249152261763811, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9976764917373657, "reward_total_composite_std": 0.004249152261763811} {"timestamp_utc": "2026-04-12T02:12:41Z", "mode": "train", "global_step": 2507, "epoch": 0.1006948628348797, "loss": 0.023, "grad_norm": 9.360161781311035, "learning_rate": 2.4060606060606062e-06, "num_tokens": 5676923.0, "completions/mean_length": 62.625, "completions/min_length": 60.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.625, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9846850633621216, "rewards/meter/std": 0.005829821340739727, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9846850633621216, "rewards/total_composite/std": 0.005829821340739727, "reward": 0.9846850633621216, "reward_std": 0.0058298311196267605, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.048959724605083466, "sampling/sampling_logp_difference/max": 4.749904632568359, "sampling/importance_sampling_ratio/min": 0.00865252036601305, "sampling/importance_sampling_ratio/mean": 0.9932248592376709, "sampling/importance_sampling_ratio/max": 1.7164337635040283, "entropy": 0.11215901840478182, "clip_ratio/low_mean": 0.015786210540682077, "clip_ratio/low_min": 0.015786210540682077, "clip_ratio/high_mean": 0.008196721319109201, "clip_ratio/high_max": 0.008196721319109201, "clip_ratio/region_mean": 0.02398293185979128, "reward_total_mean": 0.9846850633621216, "reward_meter_mean": 0.9846850633621216, "reward_meter_std": 0.005829821340739727, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9846850633621216, "reward_total_composite_std": 0.005829821340739727} {"timestamp_utc": "2026-04-12T02:12:46Z", "mode": "train", "global_step": 2508, "epoch": 0.10073502831666466, "loss": 0.0105, "grad_norm": 2.719905376434326, "learning_rate": 2.403030303030303e-06, "num_tokens": 5678652.0, "completions/mean_length": 61.125, "completions/min_length": 59.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.125, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9946359992027283, "rewards/meter/std": 0.0007906173705123365, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9946359992027283, "rewards/total_composite/std": 0.0007906173705123365, "reward": 0.9946359992027283, "reward_std": 0.0007906315731815994, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019306229427456856, "sampling/sampling_logp_difference/max": 1.058699131011963, "sampling/importance_sampling_ratio/min": 0.3469068109989166, "sampling/importance_sampling_ratio/mean": 1.002495288848877, "sampling/importance_sampling_ratio/max": 1.7869699001312256, "entropy": 0.10735112428665161, "clip_ratio/low_mean": 0.010080644860863686, "clip_ratio/low_min": 0.010080644860863686, "clip_ratio/high_mean": 0.010315365390852094, "clip_ratio/high_max": 0.010315365390852094, "clip_ratio/region_mean": 0.02039601025171578, "reward_total_mean": 0.9946359992027283, "reward_meter_mean": 0.9946359992027283, "reward_meter_std": 0.0007906173705123365, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9946359992027283, "reward_total_composite_std": 0.0007906173705123365} {"timestamp_utc": "2026-04-12T02:12:50Z", "mode": "train", "global_step": 2509, "epoch": 0.10077519379844961, "loss": -0.0075, "grad_norm": 5.611873626708984, "learning_rate": 2.4000000000000003e-06, "num_tokens": 5680475.0, "completions/mean_length": 66.875, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.875, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9922016859054565, "rewards/meter/std": 0.016830753535032272, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9922016859054565, "rewards/total_composite/std": 0.016830753535032272, "reward": 0.9922016859054565, "reward_std": 0.016830753535032272, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006449920125305653, "sampling/sampling_logp_difference/max": 1.5570316314697266, "sampling/importance_sampling_ratio/min": 0.21076074242591858, "sampling/importance_sampling_ratio/mean": 0.9988427758216858, "sampling/importance_sampling_ratio/max": 1.0730388164520264, "entropy": 0.01955289370380342, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0018939394503831863, "reward_total_mean": 0.9922016859054565, "reward_meter_mean": 0.9922016859054565, "reward_meter_std": 0.016830753535032272, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9922016859054565, "reward_total_composite_std": 0.016830753535032272} {"timestamp_utc": "2026-04-12T02:12:55Z", "mode": "train", "global_step": 2510, "epoch": 0.10081535928023456, "loss": -0.0001, "grad_norm": 0.017269477248191833, "learning_rate": 2.396969696969697e-06, "num_tokens": 5682435.0, "completions/mean_length": 67.0, "completions/min_length": 67.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981517195701599, "rewards/meter/std": 1.4962344039304298e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981517195701599, "rewards/total_composite/std": 1.4962344039304298e-06, "reward": 0.9981517195701599, "reward_std": 1.5052806929816143e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0025519414339214563, "sampling/sampling_logp_difference/max": 0.5824198722839355, "sampling/importance_sampling_ratio/min": 0.5585451126098633, "sampling/importance_sampling_ratio/mean": 1.0006331205368042, "sampling/importance_sampling_ratio/max": 1.0855460166931152, "entropy": 0.014974120538681746, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9981517195701599, "reward_meter_mean": 0.9981517195701599, "reward_meter_std": 1.4962344039304298e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981517195701599, "reward_total_composite_std": 1.4962344039304298e-06} {"timestamp_utc": "2026-04-12T02:13:00Z", "mode": "train", "global_step": 2511, "epoch": 0.10085552476201952, "loss": 0.0007, "grad_norm": 0.2027793526649475, "learning_rate": 2.393939393939394e-06, "num_tokens": 5684525.0, "completions/mean_length": 98.25, "completions/min_length": 98.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.25, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.9980236291885376, "rewards/meter/std": 2.7453183065517806e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980236291885376, "rewards/total_composite/std": 2.7453183065517806e-05, "reward": 0.9980236291885376, "reward_std": 2.745408346527256e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009353289380669594, "sampling/sampling_logp_difference/max": 1.3464837074279785, "sampling/importance_sampling_ratio/min": 0.2601534426212311, "sampling/importance_sampling_ratio/mean": 1.0008984804153442, "sampling/importance_sampling_ratio/max": 1.3208497762680054, "entropy": 0.0416307682171464, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0025510203558951616, "clip_ratio/high_max": 0.0025510203558951616, "clip_ratio/region_mean": 0.0025510203558951616, "reward_total_mean": 0.9980236291885376, "reward_meter_mean": 0.9980236291885376, "reward_meter_std": 2.7453183065517806e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9980236291885376, "reward_total_composite_std": 2.7453183065517806e-05} {"timestamp_utc": "2026-04-12T02:13:04Z", "mode": "train", "global_step": 2512, "epoch": 0.10089569024380447, "loss": -0.0135, "grad_norm": 4.541382312774658, "learning_rate": 2.3909090909090912e-06, "num_tokens": 5686253.0, "completions/mean_length": 61.0, "completions/min_length": 59.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9948880672454834, "rewards/meter/std": 0.0007434968720190227, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8703480958938599, "rewards/total_composite/std": 0.35167405009269714, "reward": 0.8703480958938599, "reward_std": 0.35167405009269714, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0175700094550848, "sampling/sampling_logp_difference/max": 0.7211494445800781, "sampling/importance_sampling_ratio/min": 0.48619309067726135, "sampling/importance_sampling_ratio/mean": 0.9973222613334656, "sampling/importance_sampling_ratio/max": 1.566518783569336, "entropy": 0.11922997329384089, "clip_ratio/low_mean": 0.0042372881434857845, "clip_ratio/low_min": 0.0042372881434857845, "clip_ratio/high_mean": 0.020458750892430544, "clip_ratio/high_max": 0.020458750892430544, "clip_ratio/region_mean": 0.02469603903591633, "reward_total_mean": 0.8703480958938599, "reward_meter_mean": 0.9948880672454834, "reward_meter_std": 0.0007434968720190227, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8703480958938599, "reward_total_composite_std": 0.35167405009269714} {"timestamp_utc": "2026-04-12T02:13:10Z", "mode": "train", "global_step": 2513, "epoch": 0.10093585572558943, "loss": 0.0078, "grad_norm": 3.2557570934295654, "learning_rate": 2.387878787878788e-06, "num_tokens": 5689270.0, "completions/mean_length": 166.125, "completions/min_length": 159.0, "completions/max_length": 173.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 166.125, "completions/min_terminated_length": 159.0, "completions/max_terminated_length": 173.0, "rewards/meter/mean": 0.9812671542167664, "rewards/meter/std": 0.015679193660616875, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.930555522441864, "rewards/repeat_penalty/std": 0.08266931027173996, "rewards/total_composite/mean": 0.7611844539642334, "rewards/total_composite/std": 0.07191365212202072, "reward": 0.7611844539642334, "reward_std": 0.07191364467144012, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04680168256163597, "sampling/sampling_logp_difference/max": 1.3245468139648438, "sampling/importance_sampling_ratio/min": 0.26592347025871277, "sampling/importance_sampling_ratio/mean": 1.0082600116729736, "sampling/importance_sampling_ratio/max": 1.982582449913025, "entropy": 0.37756787799298763, "clip_ratio/low_mean": 0.015607842477038503, "clip_ratio/low_min": 0.015607842477038503, "clip_ratio/high_mean": 0.014292625011876225, "clip_ratio/high_max": 0.014292625011876225, "clip_ratio/region_mean": 0.029900467488914728, "reward_total_mean": 0.7611844539642334, "reward_meter_mean": 0.9812671542167664, "reward_meter_std": 0.015679193660616875, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.930555522441864, "reward_repeat_penalty_std": 0.08266931027173996, "reward_total_composite_mean": 0.7611844539642334, "reward_total_composite_std": 0.07191365212202072} {"timestamp_utc": "2026-04-12T02:13:15Z", "mode": "train", "global_step": 2514, "epoch": 0.10097602120737438, "loss": 0.0002, "grad_norm": 0.11680782586336136, "learning_rate": 2.3848484848484853e-06, "num_tokens": 5691062.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.999394416809082, "rewards/meter/std": 3.964109055232257e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999394416809082, "rewards/total_composite/std": 3.964109055232257e-06, "reward": 0.999394416809082, "reward_std": 3.96097129851114e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004099253565073013, "sampling/sampling_logp_difference/max": 0.47015440464019775, "sampling/importance_sampling_ratio/min": 0.6249058246612549, "sampling/importance_sampling_ratio/mean": 1.0025361776351929, "sampling/importance_sampling_ratio/max": 1.4564316272735596, "entropy": 0.017428745748475194, "clip_ratio/low_mean": 0.001953125, "clip_ratio/low_min": 0.001953125, "clip_ratio/high_mean": 0.001953125, "clip_ratio/high_max": 0.001953125, "clip_ratio/region_mean": 0.00390625, "reward_total_mean": 0.999394416809082, "reward_meter_mean": 0.999394416809082, "reward_meter_std": 3.964109055232257e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.999394416809082, "reward_total_composite_std": 3.964109055232257e-06} {"timestamp_utc": "2026-04-12T02:13:20Z", "mode": "train", "global_step": 2515, "epoch": 0.10101618668915933, "loss": 0.0025, "grad_norm": 4.214909076690674, "learning_rate": 2.381818181818182e-06, "num_tokens": 5693436.0, "completions/mean_length": 124.75, "completions/min_length": 119.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 124.75, "completions/min_terminated_length": 119.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.9937890768051147, "rewards/meter/std": 0.000979111879132688, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9285714626312256, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.9227913618087769, "rewards/total_composite/std": 0.07571373134851456, "reward": 0.9227913618087769, "reward_std": 0.07571373879909515, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021321963518857956, "sampling/sampling_logp_difference/max": 1.0132427215576172, "sampling/importance_sampling_ratio/min": 0.36303985118865967, "sampling/importance_sampling_ratio/mean": 1.0026657581329346, "sampling/importance_sampling_ratio/max": 1.6431286334991455, "entropy": 0.14727903716266155, "clip_ratio/low_mean": 0.0070080646546557546, "clip_ratio/low_min": 0.0070080646546557546, "clip_ratio/high_mean": 0.009006721433252096, "clip_ratio/high_max": 0.009006721433252096, "clip_ratio/region_mean": 0.01601478608790785, "reward_total_mean": 0.9227913618087769, "reward_meter_mean": 0.9937890768051147, "reward_meter_std": 0.000979111879132688, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9285714626312256, "reward_repeat_penalty_std": 0.07636035233736038, "reward_total_composite_mean": 0.9227913618087769, "reward_total_composite_std": 0.07571373134851456} {"timestamp_utc": "2026-04-12T02:13:25Z", "mode": "train", "global_step": 2516, "epoch": 0.10105635217094429, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.378787878787879e-06, "num_tokens": 5695612.0, "completions/mean_length": 106.0, "completions/min_length": 106.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.0, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.6869902610778809, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5888487696647644, "rewards/total_composite/std": 0.0, "reward": 0.5888487696647644, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0005320468917489052, "sampling/sampling_logp_difference/max": 0.00885598175227642, "sampling/importance_sampling_ratio/min": 0.9996694326400757, "sampling/importance_sampling_ratio/mean": 1.0005320310592651, "sampling/importance_sampling_ratio/max": 1.0088952779769897, "entropy": 0.0042294226586818695, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.5888487696647644, "reward_meter_mean": 0.6869902610778809, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5888487696647644, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:13:30Z", "mode": "train", "global_step": 2517, "epoch": 0.10109651765272924, "loss": 0.0787, "grad_norm": 6.815549850463867, "learning_rate": 2.375757575757576e-06, "num_tokens": 5697454.0, "completions/mean_length": 67.25, "completions/min_length": 59.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.25, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.7423630952835083, "rewards/meter/std": 0.3543585538864136, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9249999523162842, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.6705501079559326, "rewards/total_composite/std": 0.3163512647151947, "reward": 0.6705501079559326, "reward_std": 0.3163512945175171, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07449356466531754, "sampling/sampling_logp_difference/max": 1.8357524871826172, "sampling/importance_sampling_ratio/min": 0.15949344635009766, "sampling/importance_sampling_ratio/mean": 1.0203852653503418, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45786258578300476, "clip_ratio/low_mean": 0.02324857749044895, "clip_ratio/low_min": 0.02324857749044895, "clip_ratio/high_mean": 0.0541982427239418, "clip_ratio/high_max": 0.0541982427239418, "clip_ratio/region_mean": 0.07744682021439075, "reward_total_mean": 0.6705501079559326, "reward_meter_mean": 0.7423630952835083, "reward_meter_std": 0.3543585538864136, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9249999523162842, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.6705501079559326, "reward_total_composite_std": 0.3163512647151947} {"timestamp_utc": "2026-04-12T02:13:34Z", "mode": "train", "global_step": 2518, "epoch": 0.1011366831345142, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.372727272727273e-06, "num_tokens": 5698926.0, "completions/mean_length": 29.0, "completions/min_length": 29.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9957436919212341, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957436919212341, "rewards/total_composite/std": 0.0, "reward": 0.9957436919212341, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0005153539241291583, "sampling/sampling_logp_difference/max": 0.0398876816034317, "sampling/importance_sampling_ratio/min": 0.9608973860740662, "sampling/importance_sampling_ratio/mean": 0.9997519254684448, "sampling/importance_sampling_ratio/max": 1.0032331943511963, "entropy": 0.007087996578775346, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9957436919212341, "reward_meter_mean": 0.9957436919212341, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9957436919212341, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:13:39Z", "mode": "train", "global_step": 2519, "epoch": 0.10117684861629915, "loss": -0.0053, "grad_norm": 1.6101531982421875, "learning_rate": 2.36969696969697e-06, "num_tokens": 5701507.0, "completions/mean_length": 126.625, "completions/min_length": 125.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 126.625, "completions/min_terminated_length": 125.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9869413375854492, "rewards/meter/std": 0.010035636834800243, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8928571939468384, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.8810223340988159, "rewards/total_composite/std": 0.06303300708532333, "reward": 0.8810223340988159, "reward_std": 0.06303300708532333, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012142295949161053, "sampling/sampling_logp_difference/max": 0.7789406776428223, "sampling/importance_sampling_ratio/min": 0.458891898393631, "sampling/importance_sampling_ratio/mean": 1.0051202774047852, "sampling/importance_sampling_ratio/max": 1.6338825225830078, "entropy": 0.10657109878957272, "clip_ratio/low_mean": 0.003992063691839576, "clip_ratio/low_min": 0.003992063691839576, "clip_ratio/high_mean": 0.002922117244452238, "clip_ratio/high_max": 0.002922117244452238, "clip_ratio/region_mean": 0.006914180936291814, "reward_total_mean": 0.8810223340988159, "reward_meter_mean": 0.9869413375854492, "reward_meter_std": 0.010035636834800243, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8928571939468384, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.8810223340988159, "reward_total_composite_std": 0.06303300708532333} {"timestamp_utc": "2026-04-12T02:13:44Z", "mode": "train", "global_step": 2520, "epoch": 0.1012170140980841, "loss": 0.0271, "grad_norm": 8.599313735961914, "learning_rate": 2.3666666666666667e-06, "num_tokens": 5702952.0, "completions/mean_length": 36.625, "completions/min_length": 35.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.625, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.9569580554962158, "rewards/meter/std": 0.05064847692847252, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9569580554962158, "rewards/total_composite/std": 0.05064847692847252, "reward": 0.9569580554962158, "reward_std": 0.05064847320318222, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05431085824966431, "sampling/sampling_logp_difference/max": 1.5028982162475586, "sampling/importance_sampling_ratio/min": 0.22248442471027374, "sampling/importance_sampling_ratio/mean": 1.0103925466537476, "sampling/importance_sampling_ratio/max": 1.500638484954834, "entropy": 0.44045621529221535, "clip_ratio/low_mean": 0.013246799120679498, "clip_ratio/low_min": 0.013246799120679498, "clip_ratio/high_mean": 0.02055674116127193, "clip_ratio/high_max": 0.02055674116127193, "clip_ratio/region_mean": 0.03380354028195143, "reward_total_mean": 0.9569580554962158, "reward_meter_mean": 0.9569580554962158, "reward_meter_std": 0.05064847692847252, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9569580554962158, "reward_total_composite_std": 0.05064847692847252} {"timestamp_utc": "2026-04-12T02:13:48Z", "mode": "train", "global_step": 2521, "epoch": 0.10125717957986906, "loss": 0.004, "grad_norm": 4.829682350158691, "learning_rate": 2.363636363636364e-06, "num_tokens": 5704880.0, "completions/mean_length": 61.0, "completions/min_length": 60.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.994844377040863, "rewards/meter/std": 0.00046124093933030963, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.994844377040863, "rewards/total_composite/std": 0.00046124093933030963, "reward": 0.994844377040863, "reward_std": 0.0004612513293977827, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02323114313185215, "sampling/sampling_logp_difference/max": 2.4490623474121094, "sampling/importance_sampling_ratio/min": 0.08637453615665436, "sampling/importance_sampling_ratio/mean": 0.9970283508300781, "sampling/importance_sampling_ratio/max": 1.8434381484985352, "entropy": 0.09736423194408417, "clip_ratio/low_mean": 0.0020491802133619785, "clip_ratio/low_min": 0.0020491802133619785, "clip_ratio/high_mean": 0.006147540640085936, "clip_ratio/high_max": 0.006147540640085936, "clip_ratio/region_mean": 0.008196720853447914, "reward_total_mean": 0.994844377040863, "reward_meter_mean": 0.994844377040863, "reward_meter_std": 0.00046124093933030963, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.994844377040863, "reward_total_composite_std": 0.00046124093933030963} {"timestamp_utc": "2026-04-12T02:13:53Z", "mode": "train", "global_step": 2522, "epoch": 0.10129734506165401, "loss": 0.001, "grad_norm": 0.5622905492782593, "learning_rate": 2.360606060606061e-06, "num_tokens": 5706673.0, "completions/mean_length": 54.125, "completions/min_length": 54.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.125, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.7876288890838623, "rewards/meter/std": 2.806982047331985e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7876288890838623, "rewards/total_composite/std": 2.806982047331985e-05, "reward": 0.7876288890838623, "reward_std": 2.8069800464436412e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0021415329538285732, "sampling/sampling_logp_difference/max": 0.7427225112915039, "sampling/importance_sampling_ratio/min": 0.4758167266845703, "sampling/importance_sampling_ratio/mean": 0.9992128610610962, "sampling/importance_sampling_ratio/max": 1.0073615312576294, "entropy": 0.0036355706688482314, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7876288890838623, "reward_meter_mean": 0.7876288890838623, "reward_meter_std": 2.806982047331985e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7876288890838623, "reward_total_composite_std": 2.806982047331985e-05} {"timestamp_utc": "2026-04-12T02:14:03Z", "mode": "train", "global_step": 2523, "epoch": 0.10133751054343897, "loss": 0.099, "grad_norm": 1.5801368951797485, "learning_rate": 2.3575757575757577e-06, "num_tokens": 5711794.0, "completions/mean_length": 497.125, "completions/min_length": 476.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 495.0000305175781, "completions/min_terminated_length": 476.0, "completions/max_terminated_length": 511.0, "rewards/meter/mean": 0.9986308813095093, "rewards/meter/std": 0.00048286825767718256, "rewards/count_adherence/mean": 0.8046875, "rewards/count_adherence/std": 0.022097086533904076, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9345512986183167, "rewards/repeat_penalty/std": 0.06527597457170486, "rewards/total_composite/mean": 0.751766562461853, "rewards/total_composite/std": 0.06588190048933029, "reward": 0.751766562461853, "reward_std": 0.06588190793991089, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0517297238111496, "sampling/sampling_logp_difference/max": 1.794034481048584, "sampling/importance_sampling_ratio/min": 0.16628792881965637, "sampling/importance_sampling_ratio/mean": 1.0117627382278442, "sampling/importance_sampling_ratio/max": 1.9566318988800049, "entropy": 0.43601541593670845, "clip_ratio/low_mean": 0.009364859783090651, "clip_ratio/low_min": 0.009364859783090651, "clip_ratio/high_mean": 0.020694297272711992, "clip_ratio/high_max": 0.020694297272711992, "clip_ratio/region_mean": 0.030059157055802643, "reward_total_mean": 0.751766562461853, "reward_meter_mean": 0.9986308813095093, "reward_meter_std": 0.00048286825767718256, "reward_count_adherence_mean": 0.8046875, "reward_count_adherence_std": 0.022097086533904076, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9345512986183167, "reward_repeat_penalty_std": 0.06527597457170486, "reward_total_composite_mean": 0.751766562461853, "reward_total_composite_std": 0.06588190048933029} {"timestamp_utc": "2026-04-12T02:14:08Z", "mode": "train", "global_step": 2524, "epoch": 0.10137767602522392, "loss": 0.0218, "grad_norm": 5.143811225891113, "learning_rate": 2.3545454545454545e-06, "num_tokens": 5713893.0, "completions/mean_length": 101.375, "completions/min_length": 96.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.375, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.9959712028503418, "rewards/meter/std": 0.005785571411252022, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9959712028503418, "rewards/total_composite/std": 0.005785571411252022, "reward": 0.9959712028503418, "reward_std": 0.005785576067864895, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05520302802324295, "sampling/sampling_logp_difference/max": 1.536201000213623, "sampling/importance_sampling_ratio/min": 0.23741686344146729, "sampling/importance_sampling_ratio/mean": 1.0093885660171509, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4374755807220936, "clip_ratio/low_mean": 0.008616218925453722, "clip_ratio/low_min": 0.008616218925453722, "clip_ratio/high_mean": 0.034931490663439035, "clip_ratio/high_max": 0.034931490663439035, "clip_ratio/region_mean": 0.04354770958889276, "reward_total_mean": 0.9959712028503418, "reward_meter_mean": 0.9959712028503418, "reward_meter_std": 0.005785571411252022, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9959712028503418, "reward_total_composite_std": 0.005785571411252022} {"timestamp_utc": "2026-04-12T02:14:14Z", "mode": "train", "global_step": 2525, "epoch": 0.10141784150700887, "loss": 0.0033, "grad_norm": 2.698570966720581, "learning_rate": 2.3515151515151517e-06, "num_tokens": 5717394.0, "completions/mean_length": 227.625, "completions/min_length": 220.0, "completions/max_length": 235.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 227.625, "completions/min_terminated_length": 220.0, "completions/max_terminated_length": 235.0, "rewards/meter/mean": 0.9980570673942566, "rewards/meter/std": 0.0012863370357081294, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980570673942566, "rewards/total_composite/std": 0.0012863370357081294, "reward": 0.9980570673942566, "reward_std": 0.0012863423908129334, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06125364825129509, "sampling/sampling_logp_difference/max": 1.1091375350952148, "sampling/importance_sampling_ratio/min": 0.32984334230422974, "sampling/importance_sampling_ratio/mean": 1.0095667839050293, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5780696794390678, "clip_ratio/low_mean": 0.010438508819788694, "clip_ratio/low_min": 0.010438508819788694, "clip_ratio/high_mean": 0.03135894448496401, "clip_ratio/high_max": 0.03135894448496401, "clip_ratio/region_mean": 0.04179745330475271, "reward_total_mean": 0.9980570673942566, "reward_meter_mean": 0.9980570673942566, "reward_meter_std": 0.0012863370357081294, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9980570673942566, "reward_total_composite_std": 0.0012863370357081294} {"timestamp_utc": "2026-04-12T02:14:19Z", "mode": "train", "global_step": 2526, "epoch": 0.10145800698879383, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.348484848484849e-06, "num_tokens": 5719002.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0003320652758702636, "sampling/sampling_logp_difference/max": 0.003683246672153473, "sampling/importance_sampling_ratio/min": 0.9999605417251587, "sampling/importance_sampling_ratio/mean": 1.0003321170806885, "sampling/importance_sampling_ratio/max": 1.0036901235580444, "entropy": 0.002448402636218816, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:14:23Z", "mode": "train", "global_step": 2527, "epoch": 0.10149817247057878, "loss": 0.0, "grad_norm": 0.11079243570566177, "learning_rate": 2.345454545454546e-06, "num_tokens": 5720778.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993899464607239, "rewards/meter/std": 4.941231964039616e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993899464607239, "rewards/total_composite/std": 4.941231964039616e-06, "reward": 0.9993899464607239, "reward_std": 4.955206804879708e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00580887496471405, "sampling/sampling_logp_difference/max": 0.7116183638572693, "sampling/importance_sampling_ratio/min": 0.49084919691085815, "sampling/importance_sampling_ratio/mean": 1.0004364252090454, "sampling/importance_sampling_ratio/max": 1.3528425693511963, "entropy": 0.02413589833304286, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/high_mean": 0.00390625, "clip_ratio/high_max": 0.00390625, "clip_ratio/region_mean": 0.0078125, "reward_total_mean": 0.9993899464607239, "reward_meter_mean": 0.9993899464607239, "reward_meter_std": 4.941231964039616e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993899464607239, "reward_total_composite_std": 4.941231964039616e-06} {"timestamp_utc": "2026-04-12T02:14:33Z", "mode": "train", "global_step": 2528, "epoch": 0.10153833795236374, "loss": 0.0113, "grad_norm": 1.9397426843643188, "learning_rate": 2.3424242424242427e-06, "num_tokens": 5726703.0, "completions/mean_length": 483.625, "completions/min_length": 477.0, "completions/max_length": 491.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 483.625, "completions/min_terminated_length": 477.0, "completions/max_terminated_length": 491.0, "rewards/meter/mean": 0.9855457544326782, "rewards/meter/std": 0.01022270042449236, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9074074029922485, "rewards/repeat_penalty/std": 0.059391386806964874, "rewards/total_composite/mean": 0.7824571132659912, "rewards/total_composite/std": 0.05129082873463631, "reward": 0.7824571132659912, "reward_std": 0.051290810108184814, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03337203338742256, "sampling/sampling_logp_difference/max": 1.3847050666809082, "sampling/importance_sampling_ratio/min": 0.25039762258529663, "sampling/importance_sampling_ratio/mean": 1.0071146488189697, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2987423837184906, "clip_ratio/low_mean": 0.00951099069789052, "clip_ratio/low_min": 0.00951099069789052, "clip_ratio/high_mean": 0.020238142693415284, "clip_ratio/high_max": 0.020238142693415284, "clip_ratio/region_mean": 0.029749133391305804, "reward_total_mean": 0.7824571132659912, "reward_meter_mean": 0.9855457544326782, "reward_meter_std": 0.01022270042449236, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9074074029922485, "reward_repeat_penalty_std": 0.059391386806964874, "reward_total_composite_mean": 0.7824571132659912, "reward_total_composite_std": 0.05129082873463631} {"timestamp_utc": "2026-04-12T02:14:38Z", "mode": "train", "global_step": 2529, "epoch": 0.10157850343414869, "loss": 0.0012, "grad_norm": 3.176457405090332, "learning_rate": 2.3393939393939395e-06, "num_tokens": 5728897.0, "completions/mean_length": 107.25, "completions/min_length": 104.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.25, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.9512283802032471, "rewards/meter/std": 0.047763749957084656, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9512283802032471, "rewards/total_composite/std": 0.047763749957084656, "reward": 0.9512283802032471, "reward_std": 0.04776376485824585, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05268417298793793, "sampling/sampling_logp_difference/max": 1.3672003746032715, "sampling/importance_sampling_ratio/min": 0.2548193633556366, "sampling/importance_sampling_ratio/mean": 1.0175414085388184, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5495112352073193, "clip_ratio/low_mean": 0.015321710845455527, "clip_ratio/low_min": 0.015321710845455527, "clip_ratio/high_mean": 0.032445638440549374, "clip_ratio/high_max": 0.032445638440549374, "clip_ratio/region_mean": 0.0477673492860049, "reward_total_mean": 0.9512283802032471, "reward_meter_mean": 0.9512283802032471, "reward_meter_std": 0.047763749957084656, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9512283802032471, "reward_total_composite_std": 0.047763749957084656} {"timestamp_utc": "2026-04-12T02:14:43Z", "mode": "train", "global_step": 2530, "epoch": 0.10161866891593364, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.3363636363636367e-06, "num_tokens": 5730521.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00033885284210555255, "sampling/sampling_logp_difference/max": 0.004536015447229147, "sampling/importance_sampling_ratio/min": 0.9995154738426208, "sampling/importance_sampling_ratio/mean": 1.0003368854522705, "sampling/importance_sampling_ratio/max": 1.0045462846755981, "entropy": 0.002742367796599865, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:14:47Z", "mode": "train", "global_step": 2531, "epoch": 0.1016588343977186, "loss": -0.0018, "grad_norm": 3.0979435443878174, "learning_rate": 2.3333333333333336e-06, "num_tokens": 5732411.0, "completions/mean_length": 79.25, "completions/min_length": 77.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.25, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9984360933303833, "rewards/meter/std": 0.00047055029426701367, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9984360933303833, "rewards/total_composite/std": 0.00047055029426701367, "reward": 0.9984360933303833, "reward_std": 0.00047055797767825425, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04553651064634323, "sampling/sampling_logp_difference/max": 1.1313624382019043, "sampling/importance_sampling_ratio/min": 0.32259345054626465, "sampling/importance_sampling_ratio/mean": 1.0040619373321533, "sampling/importance_sampling_ratio/max": 1.7651348114013672, "entropy": 0.3967149890959263, "clip_ratio/low_mean": 0.01588775822892785, "clip_ratio/low_min": 0.01588775822892785, "clip_ratio/high_mean": 0.01107846642844379, "clip_ratio/high_max": 0.01107846642844379, "clip_ratio/region_mean": 0.02696622465737164, "reward_total_mean": 0.9984360933303833, "reward_meter_mean": 0.9984360933303833, "reward_meter_std": 0.00047055029426701367, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9984360933303833, "reward_total_composite_std": 0.00047055029426701367} {"timestamp_utc": "2026-04-12T02:14:52Z", "mode": "train", "global_step": 2532, "epoch": 0.10169899987950355, "loss": 0.0168, "grad_norm": 9.73486614227295, "learning_rate": 2.3303030303030304e-06, "num_tokens": 5734224.0, "completions/mean_length": 74.625, "completions/min_length": 63.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.625, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.3960377871990204, "rewards/meter/std": 0.33334165811538696, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.3960377871990204, "rewards/total_composite/std": 0.33334165811538696, "reward": 0.3960377871990204, "reward_std": 0.33334165811538696, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0764419287443161, "sampling/sampling_logp_difference/max": 1.598515510559082, "sampling/importance_sampling_ratio/min": 0.20219644904136658, "sampling/importance_sampling_ratio/mean": 1.011519432067871, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5188739486038685, "clip_ratio/low_mean": 0.032748116645962, "clip_ratio/low_min": 0.032748116645962, "clip_ratio/high_mean": 0.026046376209706068, "clip_ratio/high_max": 0.026046376209706068, "clip_ratio/region_mean": 0.05879449285566807, "reward_total_mean": 0.3960377871990204, "reward_meter_mean": 0.3960377871990204, "reward_meter_std": 0.33334165811538696, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.3960377871990204, "reward_total_composite_std": 0.33334165811538696} {"timestamp_utc": "2026-04-12T02:14:57Z", "mode": "train", "global_step": 2533, "epoch": 0.1017391653612885, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.3272727272727277e-06, "num_tokens": 5735968.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00032944101258181036, "sampling/sampling_logp_difference/max": 0.003968724515289068, "sampling/importance_sampling_ratio/min": 0.999470055103302, "sampling/importance_sampling_ratio/mean": 1.0003273487091064, "sampling/importance_sampling_ratio/max": 1.003976583480835, "entropy": 0.002526927593862638, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:15:01Z", "mode": "train", "global_step": 2534, "epoch": 0.10177933084307346, "loss": 0.0084, "grad_norm": 2.3901071548461914, "learning_rate": 2.3242424242424245e-06, "num_tokens": 5738092.0, "completions/mean_length": 91.5, "completions/min_length": 86.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.5, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.9918440580368042, "rewards/meter/std": 0.003950192127376795, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9669877886772156, "rewards/total_composite/std": 0.06943685561418533, "reward": 0.9669877886772156, "reward_std": 0.06943684071302414, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030601778998970985, "sampling/sampling_logp_difference/max": 0.814885139465332, "sampling/importance_sampling_ratio/min": 0.4426901638507843, "sampling/importance_sampling_ratio/mean": 1.0052556991577148, "sampling/importance_sampling_ratio/max": 1.6521724462509155, "entropy": 0.21182428859174252, "clip_ratio/low_mean": 0.0013440860202535987, "clip_ratio/low_min": 0.0013440860202535987, "clip_ratio/high_mean": 0.017688892083242536, "clip_ratio/high_max": 0.017688892083242536, "clip_ratio/region_mean": 0.019032978103496134, "reward_total_mean": 0.9669877886772156, "reward_meter_mean": 0.9918440580368042, "reward_meter_std": 0.003950192127376795, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9669877886772156, "reward_total_composite_std": 0.06943685561418533} {"timestamp_utc": "2026-04-12T02:15:12Z", "mode": "train", "global_step": 2535, "epoch": 0.10181949632485841, "loss": 0.06, "grad_norm": 1.6462563276290894, "learning_rate": 2.3212121212121213e-06, "num_tokens": 5743822.0, "completions/mean_length": 495.25, "completions/min_length": 479.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 492.857177734375, "completions/min_terminated_length": 479.0, "completions/max_terminated_length": 511.0, "rewards/meter/mean": 0.9958187341690063, "rewards/meter/std": 0.006666218861937523, "rewards/count_adherence/mean": 0.8046875, "rewards/count_adherence/std": 0.022097086533904076, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9601762294769287, "rewards/repeat_penalty/std": 0.0010186029830947518, "rewards/total_composite/mean": 0.7694027423858643, "rewards/total_composite/std": 0.0214486513286829, "reward": 0.7694027423858643, "reward_std": 0.021448642015457153, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05616435408592224, "sampling/sampling_logp_difference/max": 2.1436595916748047, "sampling/importance_sampling_ratio/min": 0.11722506582736969, "sampling/importance_sampling_ratio/mean": 1.014970302581787, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4710547477006912, "clip_ratio/low_mean": 0.006211606087163091, "clip_ratio/low_min": 0.006211606087163091, "clip_ratio/high_mean": 0.023648475529626012, "clip_ratio/high_max": 0.023648475529626012, "clip_ratio/region_mean": 0.029860081616789103, "reward_total_mean": 0.7694027423858643, "reward_meter_mean": 0.9958187341690063, "reward_meter_std": 0.006666218861937523, "reward_count_adherence_mean": 0.8046875, "reward_count_adherence_std": 0.022097086533904076, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9601762294769287, "reward_repeat_penalty_std": 0.0010186029830947518, "reward_total_composite_mean": 0.7694027423858643, "reward_total_composite_std": 0.0214486513286829} {"timestamp_utc": "2026-04-12T02:15:20Z", "mode": "train", "global_step": 2536, "epoch": 0.10185966180664337, "loss": 0.0168, "grad_norm": 3.1345582008361816, "learning_rate": 2.318181818181818e-06, "num_tokens": 5748428.0, "completions/mean_length": 334.75, "completions/min_length": 306.0, "completions/max_length": 353.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 334.75, "completions/min_terminated_length": 306.0, "completions/max_terminated_length": 353.0, "rewards/meter/mean": 0.9545778036117554, "rewards/meter/std": 0.0988415852189064, "rewards/count_adherence/mean": 0.8636363744735718, "rewards/count_adherence/std": 0.04859296977519989, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9794891476631165, "rewards/repeat_penalty/std": 0.028372056782245636, "rewards/total_composite/mean": 0.8046990633010864, "rewards/total_composite/std": 0.07172781974077225, "reward": 0.8046990633010864, "reward_std": 0.07172780483961105, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08371323347091675, "sampling/sampling_logp_difference/max": 4.665629863739014, "sampling/importance_sampling_ratio/min": 0.00941331684589386, "sampling/importance_sampling_ratio/mean": 1.0162872076034546, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.806172601878643, "clip_ratio/low_mean": 0.013427280820906162, "clip_ratio/low_min": 0.013427280820906162, "clip_ratio/high_mean": 0.052931731566786766, "clip_ratio/high_max": 0.052931731566786766, "clip_ratio/region_mean": 0.06635901238769293, "reward_total_mean": 0.8046990633010864, "reward_meter_mean": 0.9545778036117554, "reward_meter_std": 0.0988415852189064, "reward_count_adherence_mean": 0.8636363744735718, "reward_count_adherence_std": 0.04859296977519989, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9794891476631165, "reward_repeat_penalty_std": 0.028372056782245636, "reward_total_composite_mean": 0.8046990633010864, "reward_total_composite_std": 0.07172781974077225} {"timestamp_utc": "2026-04-12T02:15:25Z", "mode": "train", "global_step": 2537, "epoch": 0.10189982728842832, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.3151515151515154e-06, "num_tokens": 5750148.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00031052454141899943, "sampling/sampling_logp_difference/max": 0.0035831150598824024, "sampling/importance_sampling_ratio/min": 0.999996542930603, "sampling/importance_sampling_ratio/mean": 1.0003106594085693, "sampling/importance_sampling_ratio/max": 1.0035895109176636, "entropy": 0.0026191577198915184, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:15:29Z", "mode": "train", "global_step": 2538, "epoch": 0.10193999277021328, "loss": 0.0261, "grad_norm": 8.2296781539917, "learning_rate": 2.3121212121212123e-06, "num_tokens": 5751904.0, "completions/mean_length": 70.5, "completions/min_length": 67.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.5, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9975142478942871, "rewards/meter/std": 0.003199664643034339, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9975142478942871, "rewards/total_composite/std": 0.003199664643034339, "reward": 0.9975142478942871, "reward_std": 0.0031996748875826597, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06007351353764534, "sampling/sampling_logp_difference/max": 1.1276092529296875, "sampling/importance_sampling_ratio/min": 0.3238064646720886, "sampling/importance_sampling_ratio/mean": 1.0132718086242676, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.577742449939251, "clip_ratio/low_mean": 0.02077368483878672, "clip_ratio/low_min": 0.02077368483878672, "clip_ratio/high_mean": 0.027109916205517948, "clip_ratio/high_max": 0.027109916205517948, "clip_ratio/region_mean": 0.04788360104430467, "reward_total_mean": 0.9975142478942871, "reward_meter_mean": 0.9975142478942871, "reward_meter_std": 0.003199664643034339, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9975142478942871, "reward_total_composite_std": 0.003199664643034339} {"timestamp_utc": "2026-04-12T02:15:34Z", "mode": "train", "global_step": 2539, "epoch": 0.10198015825199823, "loss": -0.0026, "grad_norm": 3.5843658447265625, "learning_rate": 2.309090909090909e-06, "num_tokens": 5753708.0, "completions/mean_length": 72.5, "completions/min_length": 71.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.5, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9982837438583374, "rewards/meter/std": 0.0006137907621450722, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9982837438583374, "rewards/total_composite/std": 0.0006137907621450722, "reward": 0.9982837438583374, "reward_std": 0.0006137940799817443, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023249905556440353, "sampling/sampling_logp_difference/max": 1.0794258117675781, "sampling/importance_sampling_ratio/min": 0.33979055285453796, "sampling/importance_sampling_ratio/mean": 1.0038199424743652, "sampling/importance_sampling_ratio/max": 1.6161738634109497, "entropy": 0.16330011747777462, "clip_ratio/low_mean": 0.005116959102451801, "clip_ratio/low_min": 0.005116959102451801, "clip_ratio/high_mean": 0.017147069913335145, "clip_ratio/high_max": 0.017147069913335145, "clip_ratio/region_mean": 0.022264029015786946, "reward_total_mean": 0.9982837438583374, "reward_meter_mean": 0.9982837438583374, "reward_meter_std": 0.0006137907621450722, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9982837438583374, "reward_total_composite_std": 0.0006137907621450722} {"timestamp_utc": "2026-04-12T02:15:41Z", "mode": "train", "global_step": 2540, "epoch": 0.10202032373378318, "loss": -0.0035, "grad_norm": 2.297759771347046, "learning_rate": 2.306060606060606e-06, "num_tokens": 5756765.0, "completions/mean_length": 193.125, "completions/min_length": 184.0, "completions/max_length": 203.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 193.125, "completions/min_terminated_length": 184.0, "completions/max_terminated_length": 203.0, "rewards/meter/mean": 0.9978808760643005, "rewards/meter/std": 0.001249481923878193, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978808760643005, "rewards/total_composite/std": 0.001249481923878193, "reward": 0.9978808760643005, "reward_std": 0.0012494820402935147, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05693472549319267, "sampling/sampling_logp_difference/max": 1.3195972442626953, "sampling/importance_sampling_ratio/min": 0.2672429084777832, "sampling/importance_sampling_ratio/mean": 1.0168569087982178, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5925077982246876, "clip_ratio/low_mean": 0.024065676843747497, "clip_ratio/low_min": 0.024065676843747497, "clip_ratio/high_mean": 0.017286017769947648, "clip_ratio/high_max": 0.017286017769947648, "clip_ratio/region_mean": 0.041351694613695145, "reward_total_mean": 0.9978808760643005, "reward_meter_mean": 0.9978808760643005, "reward_meter_std": 0.001249481923878193, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9978808760643005, "reward_total_composite_std": 0.001249481923878193} {"timestamp_utc": "2026-04-12T02:15:46Z", "mode": "train", "global_step": 2541, "epoch": 0.10206048921556814, "loss": 0.0017, "grad_norm": 5.599558353424072, "learning_rate": 2.303030303030303e-06, "num_tokens": 5758973.0, "completions/mean_length": 107.0, "completions/min_length": 105.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.0, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.9983425140380859, "rewards/meter/std": 0.0005217316211201251, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9733964800834656, "rewards/total_composite/std": 0.07076215744018555, "reward": 0.9733964800834656, "reward_std": 0.07076217234134674, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02940228208899498, "sampling/sampling_logp_difference/max": 1.1934094429016113, "sampling/importance_sampling_ratio/min": 0.3031857907772064, "sampling/importance_sampling_ratio/mean": 1.00425386428833, "sampling/importance_sampling_ratio/max": 1.613302230834961, "entropy": 0.19336522929370403, "clip_ratio/low_mean": 0.002358490601181984, "clip_ratio/low_min": 0.002358490601181984, "clip_ratio/high_mean": 0.025725040584802628, "clip_ratio/high_max": 0.025725040584802628, "clip_ratio/region_mean": 0.02808353118598461, "reward_total_mean": 0.9733964800834656, "reward_meter_mean": 0.9983425140380859, "reward_meter_std": 0.0005217316211201251, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9733964800834656, "reward_total_composite_std": 0.07076215744018555} {"timestamp_utc": "2026-04-12T02:15:50Z", "mode": "train", "global_step": 2542, "epoch": 0.10210065469735309, "loss": 0.0001, "grad_norm": 0.0943424329161644, "learning_rate": 2.3000000000000004e-06, "num_tokens": 5760901.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993921518325806, "rewards/meter/std": 5.212498763285112e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993921518325806, "rewards/total_composite/std": 5.212498763285112e-06, "reward": 0.9993921518325806, "reward_std": 5.226743269304279e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004389709793031216, "sampling/sampling_logp_difference/max": 0.464327335357666, "sampling/importance_sampling_ratio/min": 0.777456521987915, "sampling/importance_sampling_ratio/mean": 1.0001180171966553, "sampling/importance_sampling_ratio/max": 1.5909435749053955, "entropy": 0.02692276705056429, "clip_ratio/low_mean": 0.001953125, "clip_ratio/low_min": 0.001953125, "clip_ratio/high_mean": 0.0078125, "clip_ratio/high_max": 0.0078125, "clip_ratio/region_mean": 0.009765625, "reward_total_mean": 0.9993921518325806, "reward_meter_mean": 0.9993921518325806, "reward_meter_std": 5.212498763285112e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993921518325806, "reward_total_composite_std": 5.212498763285112e-06} {"timestamp_utc": "2026-04-12T02:15:58Z", "mode": "train", "global_step": 2543, "epoch": 0.10214082017913804, "loss": 0.0244, "grad_norm": 2.6460602283477783, "learning_rate": 2.2969696969696973e-06, "num_tokens": 5765713.0, "completions/mean_length": 359.5, "completions/min_length": 338.0, "completions/max_length": 384.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 359.5, "completions/min_terminated_length": 338.0, "completions/max_terminated_length": 384.0, "rewards/meter/mean": 0.9972070455551147, "rewards/meter/std": 0.0018466432811692357, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.05175493285059929, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.9864766001701355, "rewards/repeat_penalty/std": 0.025052649900317192, "rewards/total_composite/mean": 0.7973617315292358, "rewards/total_composite/std": 0.32495424151420593, "reward": 0.7973617315292358, "reward_std": 0.32495421171188354, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07025956362485886, "sampling/sampling_logp_difference/max": 1.513051986694336, "sampling/importance_sampling_ratio/min": 0.22023677825927734, "sampling/importance_sampling_ratio/mean": 1.0157252550125122, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6531303524971008, "clip_ratio/low_mean": 0.0061848959885537624, "clip_ratio/low_min": 0.0061848959885537624, "clip_ratio/high_mean": 0.03669871832244098, "clip_ratio/high_max": 0.03669871832244098, "clip_ratio/region_mean": 0.042883614310994744, "reward_total_mean": 0.7973617315292358, "reward_meter_mean": 0.9972070455551147, "reward_meter_std": 0.0018466432811692357, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.05175493285059929, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.9864766001701355, "reward_repeat_penalty_std": 0.025052649900317192, "reward_total_composite_mean": 0.7973617315292358, "reward_total_composite_std": 0.32495424151420593} {"timestamp_utc": "2026-04-12T02:16:03Z", "mode": "train", "global_step": 2544, "epoch": 0.102180985660923, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.293939393939394e-06, "num_tokens": 5767393.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0003962568298447877, "sampling/sampling_logp_difference/max": 0.008148963563144207, "sampling/importance_sampling_ratio/min": 0.9980019927024841, "sampling/importance_sampling_ratio/mean": 1.000386118888855, "sampling/importance_sampling_ratio/max": 1.0081822872161865, "entropy": 0.0035496088094078004, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:16:07Z", "mode": "train", "global_step": 2545, "epoch": 0.10222115114270795, "loss": -0.0032, "grad_norm": 3.4628183841705322, "learning_rate": 2.2909090909090913e-06, "num_tokens": 5769250.0, "completions/mean_length": 61.125, "completions/min_length": 60.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.125, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9939138889312744, "rewards/meter/std": 0.0031664494890719652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9939138889312744, "rewards/total_composite/std": 0.0031664494890719652, "reward": 0.9939138889312744, "reward_std": 0.0031664466951042414, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020268075168132782, "sampling/sampling_logp_difference/max": 1.9380130767822266, "sampling/importance_sampling_ratio/min": 0.14398977160453796, "sampling/importance_sampling_ratio/mean": 1.004980444908142, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09053080063313246, "clip_ratio/low_mean": 0.0020491802133619785, "clip_ratio/low_min": 0.0020491802133619785, "clip_ratio/high_mean": 0.004065309185534716, "clip_ratio/high_max": 0.004065309185534716, "clip_ratio/region_mean": 0.006114489398896694, "reward_total_mean": 0.9939138889312744, "reward_meter_mean": 0.9939138889312744, "reward_meter_std": 0.0031664494890719652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9939138889312744, "reward_total_composite_std": 0.0031664494890719652} {"timestamp_utc": "2026-04-12T02:16:12Z", "mode": "train", "global_step": 2546, "epoch": 0.1022613166244929, "loss": -0.0005, "grad_norm": 0.5305520296096802, "learning_rate": 2.287878787878788e-06, "num_tokens": 5771130.0, "completions/mean_length": 67.0, "completions/min_length": 67.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981376528739929, "rewards/meter/std": 2.376116935920436e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981376528739929, "rewards/total_composite/std": 2.376116935920436e-05, "reward": 0.9981376528739929, "reward_std": 2.3771623091306537e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004880668129771948, "sampling/sampling_logp_difference/max": 0.47992897033691406, "sampling/importance_sampling_ratio/min": 0.6188273429870605, "sampling/importance_sampling_ratio/mean": 0.9998449087142944, "sampling/importance_sampling_ratio/max": 1.1664998531341553, "entropy": 0.025023984955623746, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0018656715983524919, "reward_total_mean": 0.9981376528739929, "reward_meter_mean": 0.9981376528739929, "reward_meter_std": 2.376116935920436e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981376528739929, "reward_total_composite_std": 2.376116935920436e-05} {"timestamp_utc": "2026-04-12T02:16:19Z", "mode": "train", "global_step": 2547, "epoch": 0.10230148210627786, "loss": 0.0142, "grad_norm": 3.3049399852752686, "learning_rate": 2.284848484848485e-06, "num_tokens": 5775349.0, "completions/mean_length": 322.375, "completions/min_length": 287.0, "completions/max_length": 343.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 322.375, "completions/min_terminated_length": 287.0, "completions/max_terminated_length": 343.0, "rewards/meter/mean": 0.9897097945213318, "rewards/meter/std": 0.02135160192847252, "rewards/count_adherence/mean": 0.8020833134651184, "rewards/count_adherence/std": 0.0431290864944458, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6906000375747681, "rewards/total_composite/std": 0.2835569977760315, "reward": 0.6906000375747681, "reward_std": 0.2835569977760315, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09224426746368408, "sampling/sampling_logp_difference/max": 1.8807573318481445, "sampling/importance_sampling_ratio/min": 0.15247458219528198, "sampling/importance_sampling_ratio/mean": 1.0231871604919434, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9364352449774742, "clip_ratio/low_mean": 0.008633634075522423, "clip_ratio/low_min": 0.008633634075522423, "clip_ratio/high_mean": 0.05134916538372636, "clip_ratio/high_max": 0.05134916538372636, "clip_ratio/region_mean": 0.05998279945924878, "reward_total_mean": 0.6906000375747681, "reward_meter_mean": 0.9897097945213318, "reward_meter_std": 0.02135160192847252, "reward_count_adherence_mean": 0.8020833134651184, "reward_count_adherence_std": 0.0431290864944458, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6906000375747681, "reward_total_composite_std": 0.2835569977760315} {"timestamp_utc": "2026-04-12T02:16:24Z", "mode": "train", "global_step": 2548, "epoch": 0.10234164758806281, "loss": 0.0725, "grad_norm": 13.814697265625, "learning_rate": 2.281818181818182e-06, "num_tokens": 5776993.0, "completions/mean_length": 42.5, "completions/min_length": 40.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.5, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.9308160543441772, "rewards/meter/std": 0.01748719997704029, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9308160543441772, "rewards/total_composite/std": 0.01748719997704029, "reward": 0.9308160543441772, "reward_std": 0.0174871776252985, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0993395447731018, "sampling/sampling_logp_difference/max": 2.4624288082122803, "sampling/importance_sampling_ratio/min": 0.08522769808769226, "sampling/importance_sampling_ratio/mean": 0.9965495467185974, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5177558250725269, "clip_ratio/low_mean": 0.00866977241821587, "clip_ratio/low_min": 0.00866977241821587, "clip_ratio/high_mean": 0.06528556300327182, "clip_ratio/high_max": 0.06528556300327182, "clip_ratio/region_mean": 0.07395533542148769, "reward_total_mean": 0.9308160543441772, "reward_meter_mean": 0.9308160543441772, "reward_meter_std": 0.01748719997704029, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9308160543441772, "reward_total_composite_std": 0.01748719997704029} {"timestamp_utc": "2026-04-12T02:16:30Z", "mode": "train", "global_step": 2549, "epoch": 0.10238181306984777, "loss": 0.0034, "grad_norm": 1.8126007318496704, "learning_rate": 2.278787878787879e-06, "num_tokens": 5780008.0, "completions/mean_length": 191.875, "completions/min_length": 187.0, "completions/max_length": 196.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 191.875, "completions/min_terminated_length": 187.0, "completions/max_terminated_length": 196.0, "rewards/meter/mean": 0.9984524250030518, "rewards/meter/std": 0.0005822303937748075, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9984524250030518, "rewards/total_composite/std": 0.0005822303937748075, "reward": 0.9984524250030518, "reward_std": 0.0005822345847263932, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.055033281445503235, "sampling/sampling_logp_difference/max": 1.1893186569213867, "sampling/importance_sampling_ratio/min": 0.3044286072254181, "sampling/importance_sampling_ratio/mean": 1.0145859718322754, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4880506880581379, "clip_ratio/low_mean": 0.014413215219974518, "clip_ratio/low_min": 0.014413215219974518, "clip_ratio/high_mean": 0.023415555711835623, "clip_ratio/high_max": 0.023415555711835623, "clip_ratio/region_mean": 0.03782877093181014, "reward_total_mean": 0.9984524250030518, "reward_meter_mean": 0.9984524250030518, "reward_meter_std": 0.0005822303937748075, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9984524250030518, "reward_total_composite_std": 0.0005822303937748075} {"timestamp_utc": "2026-04-12T02:16:34Z", "mode": "train", "global_step": 2550, "epoch": 0.10242197855163272, "loss": 0.0457, "grad_norm": 4.8445634841918945, "learning_rate": 2.275757575757576e-06, "num_tokens": 5781794.0, "completions/mean_length": 68.25, "completions/min_length": 64.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.25, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.980836033821106, "rewards/meter/std": 0.014489145018160343, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.980836033821106, "rewards/total_composite/std": 0.014489145018160343, "reward": 0.980836033821106, "reward_std": 0.014489145949482918, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05946962162852287, "sampling/sampling_logp_difference/max": 1.5740957260131836, "sampling/importance_sampling_ratio/min": 0.20719482004642487, "sampling/importance_sampling_ratio/mean": 1.006621241569519, "sampling/importance_sampling_ratio/max": 1.8987267017364502, "entropy": 0.5948160253465176, "clip_ratio/low_mean": 0.017614107578992844, "clip_ratio/low_min": 0.017614107578992844, "clip_ratio/high_mean": 0.020960328169167042, "clip_ratio/high_max": 0.020960328169167042, "clip_ratio/region_mean": 0.038574435748159885, "reward_total_mean": 0.980836033821106, "reward_meter_mean": 0.980836033821106, "reward_meter_std": 0.014489145018160343, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.980836033821106, "reward_total_composite_std": 0.014489145018160343} {"timestamp_utc": "2026-04-12T02:17:51Z", "mode": "eval", "global_step": 2550, "epoch": 0.10242197855163272, "eval_loss": NaN, "eval_runtime": 76.2084, "eval_samples_per_second": 1.365, "eval_steps_per_second": 0.171, "eval_num_tokens": 5781794.0, "eval_completions/mean_length": 210.5, "eval_completions/min_length": 62.92307692307692, "eval_completions/max_length": 409.0769230769231, "eval_completions/clipped_ratio": 0.009615384615384616, "eval_completions/mean_terminated_length": 207.71153963529147, "eval_completions/min_terminated_length": 62.92307692307692, "eval_completions/max_terminated_length": 399.2307692307692, "eval_rewards/meter/mean": 0.768360605606666, "eval_rewards/meter/std": 0.34562337971650636, "eval_rewards/count_adherence/mean": 0.9420714332507207, "eval_rewards/count_adherence/std": 0.08270380875239006, "eval_rewards/arabic_clean/mean": 0.9711538461538461, "eval_rewards/arabic_clean/std": 0.08158924258672275, "eval_rewards/repeat_penalty/mean": 0.9319783999369695, "eval_rewards/repeat_penalty/std": 0.09335804444092971, "eval_rewards/total_composite/mean": 0.6745542379525992, "eval_rewards/total_composite/std": 0.3481578013071647, "eval_reward": 0.6745542379525992, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.03692194346625071, "eval_sampling/sampling_logp_difference/max": 1.2396357609675481, "eval_sampling/importance_sampling_ratio/min": 0.29530371954807866, "eval_sampling/importance_sampling_ratio/mean": 1.0112521740106435, "eval_sampling/importance_sampling_ratio/max": 1.5434642296570997, "eval_entropy": 0.42294850601599765, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6745542379525992, "eval_reward_meter_mean": 0.768360605606666, "eval_reward_meter_std": 0.34562337971650636, "eval_reward_count_adherence_mean": 0.9420714332507207, "eval_reward_count_adherence_std": 0.08270380875239006, "eval_reward_arabic_clean_mean": 0.9711538461538461, "eval_reward_arabic_clean_std": 0.08158924258672275, "eval_reward_repeat_penalty_mean": 0.9319783999369695, "eval_reward_repeat_penalty_std": 0.09335804444092971, "eval_reward_total_composite_mean": 0.6745542379525992, "eval_reward_total_composite_std": 0.3481578013071647} {"timestamp_utc": "2026-04-12T02:17:59Z", "mode": "train", "global_step": 2551, "epoch": 0.10246214403341768, "loss": 0.004, "grad_norm": 1.7443753480911255, "learning_rate": 2.2727272727272728e-06, "num_tokens": 5784519.0, "completions/mean_length": 166.625, "completions/min_length": 166.0, "completions/max_length": 167.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 166.625, "completions/min_terminated_length": 166.0, "completions/max_terminated_length": 167.0, "rewards/meter/mean": 0.998813271522522, "rewards/meter/std": 0.00025403356994502246, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998813271522522, "rewards/total_composite/std": 0.00025403356994502246, "reward": 0.998813271522522, "reward_std": 0.00025402315077371895, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021267293021082878, "sampling/sampling_logp_difference/max": 1.1385552883148193, "sampling/importance_sampling_ratio/min": 0.32028141617774963, "sampling/importance_sampling_ratio/mean": 1.0034915208816528, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1492440328001976, "clip_ratio/low_mean": 0.005239521153271198, "clip_ratio/low_min": 0.005239521153271198, "clip_ratio/high_mean": 0.013518144143745303, "clip_ratio/high_max": 0.013518144143745303, "clip_ratio/region_mean": 0.0187576652970165, "reward_total_mean": 0.998813271522522, "reward_meter_mean": 0.998813271522522, "reward_meter_std": 0.00025403356994502246, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.998813271522522, "reward_total_composite_std": 0.00025403356994502246} {"timestamp_utc": "2026-04-12T02:18:04Z", "mode": "train", "global_step": 2552, "epoch": 0.10250230951520263, "loss": 0.0005, "grad_norm": 0.3295544683933258, "learning_rate": 2.2696969696969696e-06, "num_tokens": 5786536.0, "completions/mean_length": 98.125, "completions/min_length": 98.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.125, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.9980067014694214, "rewards/meter/std": 2.103978840750642e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980067014694214, "rewards/total_composite/std": 2.103978840750642e-05, "reward": 0.9980067014694214, "reward_std": 2.104529266944155e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008558080531656742, "sampling/sampling_logp_difference/max": 1.609175205230713, "sampling/importance_sampling_ratio/min": 0.20005254447460175, "sampling/importance_sampling_ratio/mean": 1.0038623809814453, "sampling/importance_sampling_ratio/max": 1.583804726600647, "entropy": 0.05678324867039919, "clip_ratio/low_mean": 0.0012755101779475808, "clip_ratio/low_min": 0.0012755101779475808, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0012755101779475808, "reward_total_mean": 0.9980067014694214, "reward_meter_mean": 0.9980067014694214, "reward_meter_std": 2.103978840750642e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9980067014694214, "reward_total_composite_std": 2.103978840750642e-05} {"timestamp_utc": "2026-04-12T02:18:09Z", "mode": "train", "global_step": 2553, "epoch": 0.10254247499698758, "loss": 0.0792, "grad_norm": 12.867728233337402, "learning_rate": 2.266666666666667e-06, "num_tokens": 5788162.0, "completions/mean_length": 45.25, "completions/min_length": 38.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.25, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.5925682783126831, "rewards/meter/std": 0.4251895844936371, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5925682783126831, "rewards/total_composite/std": 0.4251895844936371, "reward": 0.5925682783126831, "reward_std": 0.4251895546913147, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1222197413444519, "sampling/sampling_logp_difference/max": 4.768388271331787, "sampling/importance_sampling_ratio/min": 0.008494059555232525, "sampling/importance_sampling_ratio/mean": 0.9826568365097046, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5326411537826061, "clip_ratio/low_mean": 0.028055555652827024, "clip_ratio/low_min": 0.028055555652827024, "clip_ratio/high_mean": 0.08001901814714074, "clip_ratio/high_max": 0.08001901814714074, "clip_ratio/region_mean": 0.10807457379996777, "reward_total_mean": 0.5925682783126831, "reward_meter_mean": 0.5925682783126831, "reward_meter_std": 0.4251895844936371, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5925682783126831, "reward_total_composite_std": 0.4251895844936371} {"timestamp_utc": "2026-04-12T02:18:14Z", "mode": "train", "global_step": 2554, "epoch": 0.10258264047877254, "loss": -0.0001, "grad_norm": 3.188581705093384, "learning_rate": 2.2636363636363637e-06, "num_tokens": 5790271.0, "completions/mean_length": 89.625, "completions/min_length": 88.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.625, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9921834468841553, "rewards/meter/std": 0.003569816704839468, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9921834468841553, "rewards/total_composite/std": 0.003569816704839468, "reward": 0.9921834468841553, "reward_std": 0.0035698034334927797, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.032123863697052, "sampling/sampling_logp_difference/max": 0.9250731468200684, "sampling/importance_sampling_ratio/min": 0.39650243520736694, "sampling/importance_sampling_ratio/mean": 1.0064231157302856, "sampling/importance_sampling_ratio/max": 1.3844112157821655, "entropy": 0.2214542292058468, "clip_ratio/low_mean": 0.009753433521836996, "clip_ratio/low_min": 0.009753433521836996, "clip_ratio/high_mean": 0.011097930022515357, "clip_ratio/high_max": 0.011097930022515357, "clip_ratio/region_mean": 0.020851363544352353, "reward_total_mean": 0.9921834468841553, "reward_meter_mean": 0.9921834468841553, "reward_meter_std": 0.003569816704839468, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9921834468841553, "reward_total_composite_std": 0.003569816704839468} {"timestamp_utc": "2026-04-12T02:18:19Z", "mode": "train", "global_step": 2555, "epoch": 0.10262280596055749, "loss": 0.0004, "grad_norm": 0.22581298649311066, "learning_rate": 2.260606060606061e-06, "num_tokens": 5792325.0, "completions/mean_length": 98.75, "completions/min_length": 98.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.75, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.9993069171905518, "rewards/meter/std": 1.8531554815126583e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993069171905518, "rewards/total_composite/std": 1.8531554815126583e-05, "reward": 0.9993069171905518, "reward_std": 1.8528446162235923e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008686289191246033, "sampling/sampling_logp_difference/max": 0.6104934215545654, "sampling/importance_sampling_ratio/min": 0.5430828332901001, "sampling/importance_sampling_ratio/mean": 1.0017414093017578, "sampling/importance_sampling_ratio/max": 1.5615042448043823, "entropy": 0.042933082208037376, "clip_ratio/low_mean": 0.005063388962298632, "clip_ratio/low_min": 0.005063388962298632, "clip_ratio/high_mean": 0.0025510203558951616, "clip_ratio/high_max": 0.0025510203558951616, "clip_ratio/region_mean": 0.007614409318193793, "reward_total_mean": 0.9993069171905518, "reward_meter_mean": 0.9993069171905518, "reward_meter_std": 1.8531554815126583e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993069171905518, "reward_total_composite_std": 1.8531554815126583e-05} {"timestamp_utc": "2026-04-12T02:18:25Z", "mode": "train", "global_step": 2556, "epoch": 0.10266297144234245, "loss": 0.0125, "grad_norm": 2.697803020477295, "learning_rate": 2.2575757575757578e-06, "num_tokens": 5795614.0, "completions/mean_length": 199.125, "completions/min_length": 188.0, "completions/max_length": 209.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 199.125, "completions/min_terminated_length": 188.0, "completions/max_terminated_length": 209.0, "rewards/meter/mean": 0.9469772577285767, "rewards/meter/std": 0.0619339719414711, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9090909361839294, "rewards/repeat_penalty/std": 0.08416546136140823, "rewards/total_composite/mean": 0.8613824248313904, "rewards/total_composite/std": 0.10320668667554855, "reward": 0.8613824248313904, "reward_std": 0.10320668667554855, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04738626256585121, "sampling/sampling_logp_difference/max": 1.4838252067565918, "sampling/importance_sampling_ratio/min": 0.2267685979604721, "sampling/importance_sampling_ratio/mean": 1.012196660041809, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47996126115322113, "clip_ratio/low_mean": 0.008094559656456113, "clip_ratio/low_min": 0.008094559656456113, "clip_ratio/high_mean": 0.0364626687951386, "clip_ratio/high_max": 0.0364626687951386, "clip_ratio/region_mean": 0.04455722845159471, "reward_total_mean": 0.8613824248313904, "reward_meter_mean": 0.9469772577285767, "reward_meter_std": 0.0619339719414711, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9090909361839294, "reward_repeat_penalty_std": 0.08416546136140823, "reward_total_composite_mean": 0.8613824248313904, "reward_total_composite_std": 0.10320668667554855} {"timestamp_utc": "2026-04-12T02:18:30Z", "mode": "train", "global_step": 2557, "epoch": 0.1027031369241274, "loss": 0.0071, "grad_norm": 4.737760066986084, "learning_rate": 2.254545454545455e-06, "num_tokens": 5797542.0, "completions/mean_length": 68.0, "completions/min_length": 64.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.974303126335144, "rewards/meter/std": 0.020303523167967796, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.974303126335144, "rewards/total_composite/std": 0.020303523167967796, "reward": 0.974303126335144, "reward_std": 0.020303526893258095, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04797229915857315, "sampling/sampling_logp_difference/max": 1.6128530502319336, "sampling/importance_sampling_ratio/min": 0.19931812584400177, "sampling/importance_sampling_ratio/mean": 1.0142627954483032, "sampling/importance_sampling_ratio/max": 1.927941083908081, "entropy": 0.4355851337313652, "clip_ratio/low_mean": 0.01822700910270214, "clip_ratio/low_min": 0.01822700910270214, "clip_ratio/high_mean": 0.012197049451060593, "clip_ratio/high_max": 0.012197049451060593, "clip_ratio/region_mean": 0.030424058553762734, "reward_total_mean": 0.974303126335144, "reward_meter_mean": 0.974303126335144, "reward_meter_std": 0.020303523167967796, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.974303126335144, "reward_total_composite_std": 0.020303523167967796} {"timestamp_utc": "2026-04-12T02:18:36Z", "mode": "train", "global_step": 2558, "epoch": 0.10274330240591235, "loss": 0.0093, "grad_norm": 3.3955912590026855, "learning_rate": 2.251515151515152e-06, "num_tokens": 5799904.0, "completions/mean_length": 136.25, "completions/min_length": 133.0, "completions/max_length": 139.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 136.25, "completions/min_terminated_length": 133.0, "completions/max_terminated_length": 139.0, "rewards/meter/mean": 0.975925624370575, "rewards/meter/std": 0.038366202265024185, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9227904081344604, "rewards/total_composite/std": 0.0701058879494667, "reward": 0.9227904081344604, "reward_std": 0.07010588049888611, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05260032042860985, "sampling/sampling_logp_difference/max": 1.8960437774658203, "sampling/importance_sampling_ratio/min": 0.1501615196466446, "sampling/importance_sampling_ratio/mean": 1.006500005722046, "sampling/importance_sampling_ratio/max": 1.85683012008667, "entropy": 0.4538475573062897, "clip_ratio/low_mean": 0.022996979532763362, "clip_ratio/low_min": 0.022996979532763362, "clip_ratio/high_mean": 0.020191184477880597, "clip_ratio/high_max": 0.020191184477880597, "clip_ratio/region_mean": 0.04318816401064396, "reward_total_mean": 0.9227904081344604, "reward_meter_mean": 0.975925624370575, "reward_meter_std": 0.038366202265024185, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.9227904081344604, "reward_total_composite_std": 0.0701058879494667} {"timestamp_utc": "2026-04-12T02:18:40Z", "mode": "train", "global_step": 2559, "epoch": 0.10278346788769731, "loss": -0.0096, "grad_norm": 12.212129592895508, "learning_rate": 2.2484848484848487e-06, "num_tokens": 5801541.0, "completions/mean_length": 44.625, "completions/min_length": 40.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.625, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.8007308840751648, "rewards/meter/std": 0.19306603074073792, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8007308840751648, "rewards/total_composite/std": 0.19306603074073792, "reward": 0.8007308840751648, "reward_std": 0.19306600093841553, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09867885708808899, "sampling/sampling_logp_difference/max": 1.36256742477417, "sampling/importance_sampling_ratio/min": 0.25600266456604004, "sampling/importance_sampling_ratio/mean": 0.9941558837890625, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5555657297372818, "clip_ratio/low_mean": 0.030968211824074388, "clip_ratio/low_min": 0.030968211824074388, "clip_ratio/high_mean": 0.04882641462609172, "clip_ratio/high_max": 0.04882641462609172, "clip_ratio/region_mean": 0.0797946264501661, "reward_total_mean": 0.8007308840751648, "reward_meter_mean": 0.8007308840751648, "reward_meter_std": 0.19306603074073792, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8007308840751648, "reward_total_composite_std": 0.19306603074073792} {"timestamp_utc": "2026-04-12T02:18:48Z", "mode": "train", "global_step": 2560, "epoch": 0.10282363336948226, "loss": -0.0056, "grad_norm": 1.910988211631775, "learning_rate": 2.2454545454545455e-06, "num_tokens": 5806083.0, "completions/mean_length": 347.75, "completions/min_length": 337.0, "completions/max_length": 361.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 347.75, "completions/min_terminated_length": 337.0, "completions/max_terminated_length": 361.0, "rewards/meter/mean": 0.9986293315887451, "rewards/meter/std": 0.00039406708674505353, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9489378929138184, "rewards/repeat_penalty/std": 0.05824849754571915, "rewards/total_composite/mean": 0.7107186317443848, "rewards/total_composite/std": 0.043464019894599915, "reward": 0.7107186317443848, "reward_std": 0.04346400871872902, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.061371464282274246, "sampling/sampling_logp_difference/max": 14.583121299743652, "sampling/importance_sampling_ratio/min": 4.6412063170464535e-07, "sampling/importance_sampling_ratio/mean": 1.0138837099075317, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4555246904492378, "clip_ratio/low_mean": 0.014391222270205617, "clip_ratio/low_min": 0.014391222270205617, "clip_ratio/high_mean": 0.014688471332192421, "clip_ratio/high_max": 0.014688471332192421, "clip_ratio/region_mean": 0.029079693602398038, "reward_total_mean": 0.7107186317443848, "reward_meter_mean": 0.9986293315887451, "reward_meter_std": 0.00039406708674505353, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9489378929138184, "reward_repeat_penalty_std": 0.05824849754571915, "reward_total_composite_mean": 0.7107186317443848, "reward_total_composite_std": 0.043464019894599915} {"timestamp_utc": "2026-04-12T02:18:53Z", "mode": "train", "global_step": 2561, "epoch": 0.10286379885126722, "loss": -0.0045, "grad_norm": 3.3982653617858887, "learning_rate": 2.2424242424242428e-06, "num_tokens": 5807836.0, "completions/mean_length": 67.125, "completions/min_length": 65.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.125, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9521244764328003, "rewards/meter/std": 0.04941366985440254, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9521244764328003, "rewards/total_composite/std": 0.04941366985440254, "reward": 0.9521244764328003, "reward_std": 0.049413666129112244, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03345394507050514, "sampling/sampling_logp_difference/max": 2.0841522216796875, "sampling/importance_sampling_ratio/min": 0.12441255152225494, "sampling/importance_sampling_ratio/mean": 1.0056248903274536, "sampling/importance_sampling_ratio/max": 1.7526792287826538, "entropy": 0.28918380104005337, "clip_ratio/low_mean": 0.007634902372956276, "clip_ratio/low_min": 0.007634902372956276, "clip_ratio/high_mean": 0.022279977099969983, "clip_ratio/high_max": 0.022279977099969983, "clip_ratio/region_mean": 0.02991487947292626, "reward_total_mean": 0.9521244764328003, "reward_meter_mean": 0.9521244764328003, "reward_meter_std": 0.04941366985440254, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9521244764328003, "reward_total_composite_std": 0.04941366985440254} {"timestamp_utc": "2026-04-12T02:18:57Z", "mode": "train", "global_step": 2562, "epoch": 0.10290396433305217, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.2393939393939396e-06, "num_tokens": 5809388.0, "completions/mean_length": 32.0, "completions/min_length": 32.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "reward": 0.9992982149124146, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00015674103633500636, "sampling/sampling_logp_difference/max": 0.007963716052472591, "sampling/importance_sampling_ratio/min": 0.9975297451019287, "sampling/importance_sampling_ratio/mean": 1.0001332759857178, "sampling/importance_sampling_ratio/max": 1.0079954862594604, "entropy": 0.002722353514400311, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9992982149124146, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:19:02Z", "mode": "train", "global_step": 2563, "epoch": 0.10294412981483712, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.2363636363636364e-06, "num_tokens": 5811516.0, "completions/mean_length": 80.0, "completions/min_length": 80.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.0, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "reward": 0.7190229296684265, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00030092071392573416, "sampling/sampling_logp_difference/max": 0.00446719815954566, "sampling/importance_sampling_ratio/min": 0.999937891960144, "sampling/importance_sampling_ratio/mean": 1.000300645828247, "sampling/importance_sampling_ratio/max": 1.0044771432876587, "entropy": 0.0025233181950170547, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7190229296684265, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:19:06Z", "mode": "train", "global_step": 2564, "epoch": 0.10298429529662208, "loss": 0.0027, "grad_norm": 1.6162538528442383, "learning_rate": 2.2333333333333333e-06, "num_tokens": 5813045.0, "completions/mean_length": 37.125, "completions/min_length": 37.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.125, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.9993616938591003, "rewards/meter/std": 0.0003300250100437552, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993616938591003, "rewards/total_composite/std": 0.0003300250100437552, "reward": 0.9993616938591003, "reward_std": 0.00033003126736730337, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01185247115790844, "sampling/sampling_logp_difference/max": 1.2676396369934082, "sampling/importance_sampling_ratio/min": 0.28149527311325073, "sampling/importance_sampling_ratio/mean": 0.9978834986686707, "sampling/importance_sampling_ratio/max": 1.1445908546447754, "entropy": 0.059151682537049055, "clip_ratio/low_mean": 0.0033783784601837397, "clip_ratio/low_min": 0.0033783784601837397, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0033783784601837397, "reward_total_mean": 0.9993616938591003, "reward_meter_mean": 0.9993616938591003, "reward_meter_std": 0.0003300250100437552, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993616938591003, "reward_total_composite_std": 0.0003300250100437552} {"timestamp_utc": "2026-04-12T02:19:10Z", "mode": "train", "global_step": 2565, "epoch": 0.10302446077840703, "loss": -0.0009, "grad_norm": 0.15634091198444366, "learning_rate": 2.2303030303030305e-06, "num_tokens": 5814621.0, "completions/mean_length": 29.0, "completions/min_length": 29.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9957231283187866, "rewards/meter/std": 5.816265183966607e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957231283187866, "rewards/total_composite/std": 5.816265183966607e-05, "reward": 0.9957231283187866, "reward_std": 5.816265183966607e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0017511904006823897, "sampling/sampling_logp_difference/max": 0.1276249885559082, "sampling/importance_sampling_ratio/min": 0.8801833987236023, "sampling/importance_sampling_ratio/mean": 0.99995356798172, "sampling/importance_sampling_ratio/max": 1.0548250675201416, "entropy": 0.007859309727791697, "clip_ratio/low_mean": 0.004310344811528921, "clip_ratio/low_min": 0.004310344811528921, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.004310344811528921, "reward_total_mean": 0.9957231283187866, "reward_meter_mean": 0.9957231283187866, "reward_meter_std": 5.816265183966607e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9957231283187866, "reward_total_composite_std": 5.816265183966607e-05} {"timestamp_utc": "2026-04-12T02:19:15Z", "mode": "train", "global_step": 2566, "epoch": 0.10306462626019199, "loss": -0.0106, "grad_norm": 1.9842983484268188, "learning_rate": 2.2272727272727274e-06, "num_tokens": 5816563.0, "completions/mean_length": 78.75, "completions/min_length": 77.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.75, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.99830561876297, "rewards/meter/std": 0.001100848545320332, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.99830561876297, "rewards/total_composite/std": 0.001100848545320332, "reward": 0.99830561876297, "reward_std": 0.0011008477304130793, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02984953485429287, "sampling/sampling_logp_difference/max": 0.9692831039428711, "sampling/importance_sampling_ratio/min": 0.37935489416122437, "sampling/importance_sampling_ratio/mean": 1.0084333419799805, "sampling/importance_sampling_ratio/max": 1.649073600769043, "entropy": 0.22580487374216318, "clip_ratio/low_mean": 0.001623376621864736, "clip_ratio/low_min": 0.001623376621864736, "clip_ratio/high_mean": 0.014201456913724542, "clip_ratio/high_max": 0.014201456913724542, "clip_ratio/region_mean": 0.015824833535589278, "reward_total_mean": 0.99830561876297, "reward_meter_mean": 0.99830561876297, "reward_meter_std": 0.001100848545320332, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.99830561876297, "reward_total_composite_std": 0.001100848545320332} {"timestamp_utc": "2026-04-12T02:19:20Z", "mode": "train", "global_step": 2567, "epoch": 0.10310479174197694, "loss": 0.03, "grad_norm": 5.494159698486328, "learning_rate": 2.224242424242424e-06, "num_tokens": 5818390.0, "completions/mean_length": 70.375, "completions/min_length": 67.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.375, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9767506122589111, "rewards/meter/std": 0.017817148938775063, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9767506122589111, "rewards/total_composite/std": 0.017817148938775063, "reward": 0.9767506122589111, "reward_std": 0.017817148938775063, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05213453993201256, "sampling/sampling_logp_difference/max": 2.3027992248535156, "sampling/importance_sampling_ratio/min": 0.09997859597206116, "sampling/importance_sampling_ratio/mean": 1.0100852251052856, "sampling/importance_sampling_ratio/max": 1.9086910486221313, "entropy": 0.41449311561882496, "clip_ratio/low_mean": 0.015345451422035694, "clip_ratio/low_min": 0.015345451422035694, "clip_ratio/high_mean": 0.030598473269492388, "clip_ratio/high_max": 0.030598473269492388, "clip_ratio/region_mean": 0.04594392469152808, "reward_total_mean": 0.9767506122589111, "reward_meter_mean": 0.9767506122589111, "reward_meter_std": 0.017817148938775063, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9767506122589111, "reward_total_composite_std": 0.017817148938775063} {"timestamp_utc": "2026-04-12T02:19:24Z", "mode": "train", "global_step": 2568, "epoch": 0.1031449572237619, "loss": -0.0085, "grad_norm": 4.675406455993652, "learning_rate": 2.2212121212121214e-06, "num_tokens": 5820304.0, "completions/mean_length": 72.25, "completions/min_length": 68.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.25, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.842863917350769, "rewards/meter/std": 0.26622849702835083, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.842863917350769, "rewards/total_composite/std": 0.26622849702835083, "reward": 0.842863917350769, "reward_std": 0.26622846722602844, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06255535036325455, "sampling/sampling_logp_difference/max": 1.408411979675293, "sampling/importance_sampling_ratio/min": 0.2445313036441803, "sampling/importance_sampling_ratio/mean": 1.005039095878601, "sampling/importance_sampling_ratio/max": 1.7533190250396729, "entropy": 0.4912409372627735, "clip_ratio/low_mean": 0.008893084479495883, "clip_ratio/low_min": 0.008893084479495883, "clip_ratio/high_mean": 0.03215251048095524, "clip_ratio/high_max": 0.03215251048095524, "clip_ratio/region_mean": 0.041045594960451126, "reward_total_mean": 0.842863917350769, "reward_meter_mean": 0.842863917350769, "reward_meter_std": 0.26622849702835083, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.842863917350769, "reward_total_composite_std": 0.26622849702835083} {"timestamp_utc": "2026-04-12T02:19:29Z", "mode": "train", "global_step": 2569, "epoch": 0.10318512270554685, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.2181818181818187e-06, "num_tokens": 5821728.0, "completions/mean_length": 35.0, "completions/min_length": 35.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9980231523513794, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980231523513794, "rewards/total_composite/std": 0.0, "reward": 0.9980231523513794, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002904376306105405, "sampling/sampling_logp_difference/max": 0.0029266122728586197, "sampling/importance_sampling_ratio/min": 0.9983823895454407, "sampling/importance_sampling_ratio/mean": 1.000272274017334, "sampling/importance_sampling_ratio/max": 1.0029308795928955, "entropy": 0.002900286461226642, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9980231523513794, "reward_meter_mean": 0.9980231523513794, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9980231523513794, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:19:33Z", "mode": "train", "global_step": 2570, "epoch": 0.1032252881873318, "loss": 0.011, "grad_norm": 12.921791076660156, "learning_rate": 2.2151515151515155e-06, "num_tokens": 5823592.0, "completions/mean_length": 43.0, "completions/min_length": 39.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.0, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.8914755582809448, "rewards/meter/std": 0.08936496078968048, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8914755582809448, "rewards/total_composite/std": 0.08936496078968048, "reward": 0.8914755582809448, "reward_std": 0.08936495333909988, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09845032542943954, "sampling/sampling_logp_difference/max": 1.851944923400879, "sampling/importance_sampling_ratio/min": 0.15693165361881256, "sampling/importance_sampling_ratio/mean": 0.9992160201072693, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.562019981443882, "clip_ratio/low_mean": 0.035002983175218105, "clip_ratio/low_min": 0.035002983175218105, "clip_ratio/high_mean": 0.05010954383760691, "clip_ratio/high_max": 0.05010954383760691, "clip_ratio/region_mean": 0.08511252701282501, "reward_total_mean": 0.8914755582809448, "reward_meter_mean": 0.8914755582809448, "reward_meter_std": 0.08936496078968048, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8914755582809448, "reward_total_composite_std": 0.08936496078968048} {"timestamp_utc": "2026-04-12T02:19:38Z", "mode": "train", "global_step": 2571, "epoch": 0.10326545366911676, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.2121212121212124e-06, "num_tokens": 5824992.0, "completions/mean_length": 28.0, "completions/min_length": 28.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.0, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9943599104881287, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9943599104881287, "rewards/total_composite/std": 0.0, "reward": 0.9943599104881287, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00020078742818441242, "sampling/sampling_logp_difference/max": 0.0017774300649762154, "sampling/importance_sampling_ratio/min": 1.0, "sampling/importance_sampling_ratio/mean": 1.0002009868621826, "sampling/importance_sampling_ratio/max": 1.0017789602279663, "entropy": 0.0015226418327074498, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9943599104881287, "reward_meter_mean": 0.9943599104881287, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9943599104881287, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:19:42Z", "mode": "train", "global_step": 2572, "epoch": 0.10330561915090171, "loss": -0.0009, "grad_norm": 1.0375055074691772, "learning_rate": 2.209090909090909e-06, "num_tokens": 5826320.0, "completions/mean_length": 37.0, "completions/min_length": 37.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.0, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9995728731155396, "rewards/meter/std": 3.974059291067533e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9995728731155396, "rewards/total_composite/std": 3.974059291067533e-05, "reward": 0.9995728731155396, "reward_std": 3.974912760895677e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009617374278604984, "sampling/sampling_logp_difference/max": 0.6554868221282959, "sampling/importance_sampling_ratio/min": 0.5191892385482788, "sampling/importance_sampling_ratio/mean": 1.0013068914413452, "sampling/importance_sampling_ratio/max": 1.1907140016555786, "entropy": 0.054603309370577335, "clip_ratio/low_mean": 0.0033783784601837397, "clip_ratio/low_min": 0.0033783784601837397, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0033783784601837397, "reward_total_mean": 0.9995728731155396, "reward_meter_mean": 0.9995728731155396, "reward_meter_std": 3.974059291067533e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9995728731155396, "reward_total_composite_std": 3.974059291067533e-05} {"timestamp_utc": "2026-04-12T02:19:47Z", "mode": "train", "global_step": 2573, "epoch": 0.10334578463268666, "loss": 0.0037, "grad_norm": 5.262681484222412, "learning_rate": 2.2060606060606064e-06, "num_tokens": 5828013.0, "completions/mean_length": 67.625, "completions/min_length": 64.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.625, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9842870235443115, "rewards/meter/std": 0.03939536586403847, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8596113920211792, "rewards/total_composite/std": 0.34952229261398315, "reward": 0.8596113920211792, "reward_std": 0.34952226281166077, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06369344890117645, "sampling/sampling_logp_difference/max": 2.000176429748535, "sampling/importance_sampling_ratio/min": 0.13531140983104706, "sampling/importance_sampling_ratio/mean": 1.0002013444900513, "sampling/importance_sampling_ratio/max": 1.7594947814941406, "entropy": 0.5100602991878986, "clip_ratio/low_mean": 0.005514706019312143, "clip_ratio/low_min": 0.005514706019312143, "clip_ratio/high_mean": 0.04813995969016105, "clip_ratio/high_max": 0.04813995969016105, "clip_ratio/region_mean": 0.05365466570947319, "reward_total_mean": 0.8596113920211792, "reward_meter_mean": 0.9842870235443115, "reward_meter_std": 0.03939536586403847, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8596113920211792, "reward_total_composite_std": 0.34952229261398315} {"timestamp_utc": "2026-04-12T02:19:52Z", "mode": "train", "global_step": 2574, "epoch": 0.10338595011447162, "loss": -0.0004, "grad_norm": 1.3739620447158813, "learning_rate": 2.2030303030303033e-06, "num_tokens": 5829911.0, "completions/mean_length": 72.25, "completions/min_length": 71.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.25, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9990666508674622, "rewards/meter/std": 0.00011669577361317351, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990666508674622, "rewards/total_composite/std": 0.00011669577361317351, "reward": 0.9990666508674622, "reward_std": 0.00011668946535792202, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01342670526355505, "sampling/sampling_logp_difference/max": 1.0127067565917969, "sampling/importance_sampling_ratio/min": 0.3632344603538513, "sampling/importance_sampling_ratio/mean": 1.0047719478607178, "sampling/importance_sampling_ratio/max": 1.3253597021102905, "entropy": 0.09299392439424992, "clip_ratio/low_mean": 0.0017605633474886417, "clip_ratio/low_min": 0.0017605633474886417, "clip_ratio/high_mean": 0.008656773250550032, "clip_ratio/high_max": 0.008656773250550032, "clip_ratio/region_mean": 0.010417336598038673, "reward_total_mean": 0.9990666508674622, "reward_meter_mean": 0.9990666508674622, "reward_meter_std": 0.00011669577361317351, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990666508674622, "reward_total_composite_std": 0.00011669577361317351} {"timestamp_utc": "2026-04-12T02:19:57Z", "mode": "train", "global_step": 2575, "epoch": 0.10342611559625657, "loss": -0.001, "grad_norm": 0.6759248971939087, "learning_rate": 2.2e-06, "num_tokens": 5832393.0, "completions/mean_length": 132.25, "completions/min_length": 130.0, "completions/max_length": 134.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.25, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.9992468357086182, "rewards/meter/std": 8.659085870021954e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992468357086182, "rewards/total_composite/std": 8.659085870021954e-05, "reward": 0.9992468357086182, "reward_std": 8.659218292450532e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010475593619048595, "sampling/sampling_logp_difference/max": 1.35489821434021, "sampling/importance_sampling_ratio/min": 0.2579735517501831, "sampling/importance_sampling_ratio/mean": 1.0009084939956665, "sampling/importance_sampling_ratio/max": 1.3609951734542847, "entropy": 0.08119754772633314, "clip_ratio/low_mean": 0.004734848625957966, "clip_ratio/low_min": 0.004734848625957966, "clip_ratio/high_mean": 0.006622972548939288, "clip_ratio/high_max": 0.006622972548939288, "clip_ratio/region_mean": 0.011357821174897254, "reward_total_mean": 0.9992468357086182, "reward_meter_mean": 0.9992468357086182, "reward_meter_std": 8.659085870021954e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992468357086182, "reward_total_composite_std": 8.659085870021954e-05} {"timestamp_utc": "2026-04-12T02:20:02Z", "mode": "train", "global_step": 2576, "epoch": 0.10346628107804152, "loss": 0.0001, "grad_norm": 1.7544798851013184, "learning_rate": 2.196969696969697e-06, "num_tokens": 5834484.0, "completions/mean_length": 80.375, "completions/min_length": 79.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.375, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9987189769744873, "rewards/meter/std": 0.00029780645854771137, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9987189769744873, "rewards/total_composite/std": 0.00029780645854771137, "reward": 0.9987189769744873, "reward_std": 0.00029780055047012866, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023812301456928253, "sampling/sampling_logp_difference/max": 0.7268044948577881, "sampling/importance_sampling_ratio/min": 0.48345139622688293, "sampling/importance_sampling_ratio/mean": 1.005149483680725, "sampling/importance_sampling_ratio/max": 1.3658860921859741, "entropy": 0.21419258043169975, "clip_ratio/low_mean": 0.0062509768176823854, "clip_ratio/low_min": 0.0062509768176823854, "clip_ratio/high_mean": 0.014005606528371572, "clip_ratio/high_max": 0.014005606528371572, "clip_ratio/region_mean": 0.020256583346053958, "reward_total_mean": 0.9987189769744873, "reward_meter_mean": 0.9987189769744873, "reward_meter_std": 0.00029780645854771137, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9987189769744873, "reward_total_composite_std": 0.00029780645854771137} {"timestamp_utc": "2026-04-12T02:20:06Z", "mode": "train", "global_step": 2577, "epoch": 0.10350644655982648, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.193939393939394e-06, "num_tokens": 5836412.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0015179180772975087, "sampling/sampling_logp_difference/max": 0.12468904256820679, "sampling/importance_sampling_ratio/min": 0.882771372795105, "sampling/importance_sampling_ratio/mean": 1.0000942945480347, "sampling/importance_sampling_ratio/max": 1.0703659057617188, "entropy": 0.015041560400277376, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:20:11Z", "mode": "train", "global_step": 2578, "epoch": 0.10354661204161143, "loss": 0.0004, "grad_norm": 0.25849685072898865, "learning_rate": 2.190909090909091e-06, "num_tokens": 5838518.0, "completions/mean_length": 98.25, "completions/min_length": 98.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.25, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.9980080127716064, "rewards/meter/std": 1.598181188455783e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980080127716064, "rewards/total_composite/std": 1.598181188455783e-05, "reward": 0.9980080127716064, "reward_std": 1.5980675016180612e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007937485352158546, "sampling/sampling_logp_difference/max": 1.1361374855041504, "sampling/importance_sampling_ratio/min": 0.3210567235946655, "sampling/importance_sampling_ratio/mean": 1.0036529302597046, "sampling/importance_sampling_ratio/max": 1.5758888721466064, "entropy": 0.0589327453635633, "clip_ratio/low_mean": 0.0038007627008482814, "clip_ratio/low_min": 0.0038007627008482814, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0038007627008482814, "reward_total_mean": 0.9980080127716064, "reward_meter_mean": 0.9980080127716064, "reward_meter_std": 1.598181188455783e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9980080127716064, "reward_total_composite_std": 1.598181188455783e-05} {"timestamp_utc": "2026-04-12T02:20:16Z", "mode": "train", "global_step": 2579, "epoch": 0.10358677752339639, "loss": 0.0013, "grad_norm": 3.5709714889526367, "learning_rate": 2.187878787878788e-06, "num_tokens": 5840674.0, "completions/mean_length": 104.5, "completions/min_length": 98.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 104.5, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.9814869165420532, "rewards/meter/std": 0.012266860343515873, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9814869165420532, "rewards/total_composite/std": 0.012266860343515873, "reward": 0.9814869165420532, "reward_std": 0.012266861274838448, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05761587247252464, "sampling/sampling_logp_difference/max": 1.8346548080444336, "sampling/importance_sampling_ratio/min": 0.15966860949993134, "sampling/importance_sampling_ratio/mean": 1.0061341524124146, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47100016474723816, "clip_ratio/low_mean": 0.01786035276018083, "clip_ratio/low_min": 0.01786035276018083, "clip_ratio/high_mean": 0.02378739172127098, "clip_ratio/high_max": 0.02378739172127098, "clip_ratio/region_mean": 0.04164774448145181, "reward_total_mean": 0.9814869165420532, "reward_meter_mean": 0.9814869165420532, "reward_meter_std": 0.012266860343515873, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9814869165420532, "reward_total_composite_std": 0.012266860343515873} {"timestamp_utc": "2026-04-12T02:20:21Z", "mode": "train", "global_step": 2580, "epoch": 0.10362694300518134, "loss": 0.009, "grad_norm": 4.734373569488525, "learning_rate": 2.184848484848485e-06, "num_tokens": 5842442.0, "completions/mean_length": 59.0, "completions/min_length": 56.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.0, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9932022094726562, "rewards/meter/std": 0.002190361265093088, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9932022094726562, "rewards/total_composite/std": 0.002190361265093088, "reward": 0.9932022094726562, "reward_std": 0.002190381521359086, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03438284620642662, "sampling/sampling_logp_difference/max": 1.2286577224731445, "sampling/importance_sampling_ratio/min": 0.2926851809024811, "sampling/importance_sampling_ratio/mean": 1.0038007497787476, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.25036873295903206, "clip_ratio/low_mean": 0.008511104388162494, "clip_ratio/low_min": 0.008511104388162494, "clip_ratio/high_mean": 0.01871931995265186, "clip_ratio/high_max": 0.01871931995265186, "clip_ratio/region_mean": 0.027230424340814352, "reward_total_mean": 0.9932022094726562, "reward_meter_mean": 0.9932022094726562, "reward_meter_std": 0.002190361265093088, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9932022094726562, "reward_total_composite_std": 0.002190361265093088} {"timestamp_utc": "2026-04-12T02:20:25Z", "mode": "train", "global_step": 2581, "epoch": 0.1036671084869663, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.181818181818182e-06, "num_tokens": 5844282.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0005618593422695994, "sampling/sampling_logp_difference/max": 0.03692399710416794, "sampling/importance_sampling_ratio/min": 0.9637494087219238, "sampling/importance_sampling_ratio/mean": 0.9999182224273682, "sampling/importance_sampling_ratio/max": 1.0165749788284302, "entropy": 0.006089928210712969, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:20:30Z", "mode": "train", "global_step": 2582, "epoch": 0.10370727396875125, "loss": -0.0011, "grad_norm": 1.6909281015396118, "learning_rate": 2.1787878787878788e-06, "num_tokens": 5846329.0, "completions/mean_length": 72.875, "completions/min_length": 72.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.875, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9989848136901855, "rewards/meter/std": 0.00030317477649077773, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989848136901855, "rewards/total_composite/std": 0.00030317477649077773, "reward": 0.9989848136901855, "reward_std": 0.0003031773376278579, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014997649937868118, "sampling/sampling_logp_difference/max": 1.249094009399414, "sampling/importance_sampling_ratio/min": 0.2867645025253296, "sampling/importance_sampling_ratio/mean": 1.0025291442871094, "sampling/importance_sampling_ratio/max": 1.3127074241638184, "entropy": 0.0825590007007122, "clip_ratio/low_mean": 0.0017123287543654442, "clip_ratio/low_min": 0.0017123287543654442, "clip_ratio/high_mean": 0.0017361111240461469, "clip_ratio/high_max": 0.0017361111240461469, "clip_ratio/region_mean": 0.003448439878411591, "reward_total_mean": 0.9989848136901855, "reward_meter_mean": 0.9989848136901855, "reward_meter_std": 0.00030317477649077773, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9989848136901855, "reward_total_composite_std": 0.00030317477649077773} {"timestamp_utc": "2026-04-12T02:20:35Z", "mode": "train", "global_step": 2583, "epoch": 0.1037474394505362, "loss": 0.0175, "grad_norm": 4.086141109466553, "learning_rate": 2.175757575757576e-06, "num_tokens": 5848580.0, "completions/mean_length": 120.375, "completions/min_length": 116.0, "completions/max_length": 123.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.375, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.989915132522583, "rewards/meter/std": 0.006506068166345358, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9723613262176514, "rewards/total_composite/std": 0.052766378968954086, "reward": 0.9723613262176514, "reward_std": 0.05276636406779289, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.059543609619140625, "sampling/sampling_logp_difference/max": 1.5191116333007812, "sampling/importance_sampling_ratio/min": 0.21890626847743988, "sampling/importance_sampling_ratio/mean": 0.9998201727867126, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4293239116668701, "clip_ratio/low_mean": 0.005081300623714924, "clip_ratio/low_min": 0.005081300623714924, "clip_ratio/high_mean": 0.057334170676767826, "clip_ratio/high_max": 0.057334170676767826, "clip_ratio/region_mean": 0.06241547130048275, "reward_total_mean": 0.9723613262176514, "reward_meter_mean": 0.989915132522583, "reward_meter_std": 0.006506068166345358, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.9723613262176514, "reward_total_composite_std": 0.052766378968954086} {"timestamp_utc": "2026-04-12T02:20:40Z", "mode": "train", "global_step": 2584, "epoch": 0.10378760493232116, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.172727272727273e-06, "num_tokens": 5850268.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00024031523207668215, "sampling/sampling_logp_difference/max": 0.002918010577559471, "sampling/importance_sampling_ratio/min": 0.9997970461845398, "sampling/importance_sampling_ratio/mean": 1.0002394914627075, "sampling/importance_sampling_ratio/max": 1.0029222965240479, "entropy": 0.0018527817592257634, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:20:45Z", "mode": "train", "global_step": 2585, "epoch": 0.10382777041410611, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.16969696969697e-06, "num_tokens": 5852236.0, "completions/mean_length": 80.0, "completions/min_length": 80.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.0, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "reward": 0.7190229296684265, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002460446848999709, "sampling/sampling_logp_difference/max": 0.0034597045741975307, "sampling/importance_sampling_ratio/min": 0.9998869299888611, "sampling/importance_sampling_ratio/mean": 1.0002455711364746, "sampling/importance_sampling_ratio/max": 1.0034657716751099, "entropy": 0.0019878629682352766, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7190229296684265, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:20:53Z", "mode": "train", "global_step": 2586, "epoch": 0.10386793589589108, "loss": -0.0205, "grad_norm": 1.9402741193771362, "learning_rate": 2.166666666666667e-06, "num_tokens": 5856712.0, "completions/mean_length": 365.5, "completions/min_length": 342.0, "completions/max_length": 391.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 365.5, "completions/min_terminated_length": 342.0, "completions/max_terminated_length": 391.0, "rewards/meter/mean": 0.9985870122909546, "rewards/meter/std": 0.000464948097942397, "rewards/count_adherence/mean": 0.7916666269302368, "rewards/count_adherence/std": 0.04454353079199791, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9663312435150146, "rewards/repeat_penalty/std": 0.039662934839725494, "rewards/total_composite/mean": 0.7631400227546692, "rewards/total_composite/std": 0.03845896199345589, "reward": 0.7631400227546692, "reward_std": 0.03845897316932678, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05639910697937012, "sampling/sampling_logp_difference/max": 2.606635570526123, "sampling/importance_sampling_ratio/min": 0.07378236204385757, "sampling/importance_sampling_ratio/mean": 1.0101393461227417, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4673043563961983, "clip_ratio/low_mean": 0.01918130088597536, "clip_ratio/low_min": 0.01918130088597536, "clip_ratio/high_mean": 0.012192239752039313, "clip_ratio/high_max": 0.012192239752039313, "clip_ratio/region_mean": 0.031373540638014674, "reward_total_mean": 0.7631400227546692, "reward_meter_mean": 0.9985870122909546, "reward_meter_std": 0.000464948097942397, "reward_count_adherence_mean": 0.7916666269302368, "reward_count_adherence_std": 0.04454353079199791, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9663312435150146, "reward_repeat_penalty_std": 0.039662934839725494, "reward_total_composite_mean": 0.7631400227546692, "reward_total_composite_std": 0.03845896199345589} {"timestamp_utc": "2026-04-12T02:20:58Z", "mode": "train", "global_step": 2587, "epoch": 0.10390810137767603, "loss": -0.0067, "grad_norm": 6.472557067871094, "learning_rate": 2.163636363636364e-06, "num_tokens": 5858927.0, "completions/mean_length": 100.875, "completions/min_length": 97.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.875, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.979087233543396, "rewards/meter/std": 0.056263912469148636, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.979087233543396, "rewards/total_composite/std": 0.056263912469148636, "reward": 0.979087233543396, "reward_std": 0.056263916194438934, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04917147755622864, "sampling/sampling_logp_difference/max": 3.7527036666870117, "sampling/importance_sampling_ratio/min": 0.023454248905181885, "sampling/importance_sampling_ratio/mean": 1.0070948600769043, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4231472760438919, "clip_ratio/low_mean": 0.005154639016836882, "clip_ratio/low_min": 0.005154639016836882, "clip_ratio/high_mean": 0.038169488427229226, "clip_ratio/high_max": 0.038169488427229226, "clip_ratio/region_mean": 0.04332412744406611, "reward_total_mean": 0.979087233543396, "reward_meter_mean": 0.979087233543396, "reward_meter_std": 0.056263912469148636, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.979087233543396, "reward_total_composite_std": 0.056263912469148636} {"timestamp_utc": "2026-04-12T02:21:05Z", "mode": "train", "global_step": 2588, "epoch": 0.10394826685946099, "loss": 0.0085, "grad_norm": 1.698296308517456, "learning_rate": 2.1606060606060606e-06, "num_tokens": 5862067.0, "completions/mean_length": 202.5, "completions/min_length": 198.0, "completions/max_length": 205.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 202.5, "completions/min_terminated_length": 198.0, "completions/max_terminated_length": 205.0, "rewards/meter/mean": 0.9954801797866821, "rewards/meter/std": 0.008464212529361248, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9772727489471436, "rewards/repeat_penalty/std": 0.04208271950483322, "rewards/total_composite/mean": 0.9727857112884521, "rewards/total_composite/std": 0.04096178710460663, "reward": 0.9727857112884521, "reward_std": 0.04096178337931633, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02448193170130253, "sampling/sampling_logp_difference/max": 1.4660539627075195, "sampling/importance_sampling_ratio/min": 0.23083457350730896, "sampling/importance_sampling_ratio/mean": 1.0026134252548218, "sampling/importance_sampling_ratio/max": 1.8658126592636108, "entropy": 0.16349555738270283, "clip_ratio/low_mean": 0.0030577474972233176, "clip_ratio/low_min": 0.0030577474972233176, "clip_ratio/high_mean": 0.019755184242967516, "clip_ratio/high_max": 0.019755184242967516, "clip_ratio/region_mean": 0.022812931740190834, "reward_total_mean": 0.9727857112884521, "reward_meter_mean": 0.9954801797866821, "reward_meter_std": 0.008464212529361248, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9772727489471436, "reward_repeat_penalty_std": 0.04208271950483322, "reward_total_composite_mean": 0.9727857112884521, "reward_total_composite_std": 0.04096178710460663} {"timestamp_utc": "2026-04-12T02:21:09Z", "mode": "train", "global_step": 2589, "epoch": 0.10398843234124594, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.157575757575758e-06, "num_tokens": 5863619.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00021978352742735296, "sampling/sampling_logp_difference/max": 0.0021914218086749315, "sampling/importance_sampling_ratio/min": 0.9999104142189026, "sampling/importance_sampling_ratio/mean": 1.0002191066741943, "sampling/importance_sampling_ratio/max": 1.0021939277648926, "entropy": 0.0016017362504499033, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:21:14Z", "mode": "train", "global_step": 2590, "epoch": 0.1040285978230309, "loss": 0.0105, "grad_norm": 6.6089911460876465, "learning_rate": 2.1545454545454547e-06, "num_tokens": 5865559.0, "completions/mean_length": 70.5, "completions/min_length": 68.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.5, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9882650971412659, "rewards/meter/std": 0.010569445788860321, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9882650971412659, "rewards/total_composite/std": 0.010569445788860321, "reward": 0.9882650971412659, "reward_std": 0.010569446720182896, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05746662616729736, "sampling/sampling_logp_difference/max": 2.042877435684204, "sampling/importance_sampling_ratio/min": 0.12965509295463562, "sampling/importance_sampling_ratio/mean": 1.0161855220794678, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.514446884393692, "clip_ratio/low_mean": 0.013986697886139154, "clip_ratio/low_min": 0.013986697886139154, "clip_ratio/high_mean": 0.03218326787464321, "clip_ratio/high_max": 0.03218326787464321, "clip_ratio/region_mean": 0.04616996576078236, "reward_total_mean": 0.9882650971412659, "reward_meter_mean": 0.9882650971412659, "reward_meter_std": 0.010569445788860321, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9882650971412659, "reward_total_composite_std": 0.010569445788860321} {"timestamp_utc": "2026-04-12T02:21:18Z", "mode": "train", "global_step": 2591, "epoch": 0.10406876330481585, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.1515151515151515e-06, "num_tokens": 5866935.0, "completions/mean_length": 29.0, "completions/min_length": 29.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9957436919212341, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957436919212341, "rewards/total_composite/std": 0.0, "reward": 0.9957436919212341, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0003589413536246866, "sampling/sampling_logp_difference/max": 0.0062316060066223145, "sampling/importance_sampling_ratio/min": 0.9997910261154175, "sampling/importance_sampling_ratio/mean": 1.0003573894500732, "sampling/importance_sampling_ratio/max": 1.0062510967254639, "entropy": 0.002733113680733368, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9957436919212341, "reward_meter_mean": 0.9957436919212341, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9957436919212341, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:21:22Z", "mode": "train", "global_step": 2592, "epoch": 0.1041089287866008, "loss": 0.0079, "grad_norm": 5.403855800628662, "learning_rate": 2.148484848484849e-06, "num_tokens": 5868615.0, "completions/mean_length": 59.0, "completions/min_length": 57.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.0, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9917755126953125, "rewards/meter/std": 0.003381570801138878, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9917755126953125, "rewards/total_composite/std": 0.003381570801138878, "reward": 0.9917755126953125, "reward_std": 0.0033815728966146708, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03163475543260574, "sampling/sampling_logp_difference/max": 1.120924949645996, "sampling/importance_sampling_ratio/min": 0.3259781301021576, "sampling/importance_sampling_ratio/mean": 1.003715991973877, "sampling/importance_sampling_ratio/max": 1.8206579685211182, "entropy": 0.2373839858919382, "clip_ratio/low_mean": 0.014869472477585077, "clip_ratio/low_min": 0.014869472477585077, "clip_ratio/high_mean": 0.019220190355554223, "clip_ratio/high_max": 0.019220190355554223, "clip_ratio/region_mean": 0.0340896628331393, "reward_total_mean": 0.9917755126953125, "reward_meter_mean": 0.9917755126953125, "reward_meter_std": 0.003381570801138878, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9917755126953125, "reward_total_composite_std": 0.003381570801138878} {"timestamp_utc": "2026-04-12T02:21:29Z", "mode": "train", "global_step": 2593, "epoch": 0.10414909426838576, "loss": 0.0705, "grad_norm": 3.471240282058716, "learning_rate": 2.1454545454545456e-06, "num_tokens": 5871770.0, "completions/mean_length": 216.375, "completions/min_length": 197.0, "completions/max_length": 239.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 216.375, "completions/min_terminated_length": 197.0, "completions/max_terminated_length": 239.0, "rewards/meter/mean": 0.990897536277771, "rewards/meter/std": 0.006994582246989012, "rewards/count_adherence/mean": 0.9464285969734192, "rewards/count_adherence/std": 0.07393559068441391, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9820512533187866, "rewards/repeat_penalty/std": 0.033347416669130325, "rewards/total_composite/mean": 0.921344518661499, "rewards/total_composite/std": 0.0833897739648819, "reward": 0.921344518661499, "reward_std": 0.0833897814154625, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07251539081335068, "sampling/sampling_logp_difference/max": 1.7865657806396484, "sampling/importance_sampling_ratio/min": 0.16753454506397247, "sampling/importance_sampling_ratio/mean": 1.0076571702957153, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5770430602133274, "clip_ratio/low_mean": 0.0256543830037117, "clip_ratio/low_min": 0.0256543830037117, "clip_ratio/high_mean": 0.03610773291438818, "clip_ratio/high_max": 0.03610773291438818, "clip_ratio/region_mean": 0.06176211591809988, "reward_total_mean": 0.921344518661499, "reward_meter_mean": 0.990897536277771, "reward_meter_std": 0.006994582246989012, "reward_count_adherence_mean": 0.9464285969734192, "reward_count_adherence_std": 0.07393559068441391, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9820512533187866, "reward_repeat_penalty_std": 0.033347416669130325, "reward_total_composite_mean": 0.921344518661499, "reward_total_composite_std": 0.0833897739648819} {"timestamp_utc": "2026-04-12T02:21:34Z", "mode": "train", "global_step": 2594, "epoch": 0.10418925975017071, "loss": 0.0196, "grad_norm": 5.423774719238281, "learning_rate": 2.1424242424242425e-06, "num_tokens": 5873489.0, "completions/mean_length": 58.875, "completions/min_length": 57.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.875, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.992070198059082, "rewards/meter/std": 0.0043101259507238865, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.992070198059082, "rewards/total_composite/std": 0.0043101259507238865, "reward": 0.992070198059082, "reward_std": 0.004310120362788439, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03451910987496376, "sampling/sampling_logp_difference/max": 1.3982751369476318, "sampling/importance_sampling_ratio/min": 0.24702267348766327, "sampling/importance_sampling_ratio/mean": 0.9979619383811951, "sampling/importance_sampling_ratio/max": 1.3894490003585815, "entropy": 0.27704715728759766, "clip_ratio/low_mean": 0.010563877876847982, "clip_ratio/low_min": 0.010563877876847982, "clip_ratio/high_mean": 0.025721082463860512, "clip_ratio/high_max": 0.025721082463860512, "clip_ratio/region_mean": 0.036284960340708494, "reward_total_mean": 0.992070198059082, "reward_meter_mean": 0.992070198059082, "reward_meter_std": 0.0043101259507238865, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.992070198059082, "reward_total_composite_std": 0.0043101259507238865} {"timestamp_utc": "2026-04-12T02:21:41Z", "mode": "train", "global_step": 2595, "epoch": 0.10422942523195566, "loss": -0.0043, "grad_norm": 2.122164726257324, "learning_rate": 2.1393939393939393e-06, "num_tokens": 5877533.0, "completions/mean_length": 299.5, "completions/min_length": 293.0, "completions/max_length": 307.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 299.5, "completions/min_terminated_length": 293.0, "completions/max_terminated_length": 307.0, "rewards/meter/mean": 0.9690572023391724, "rewards/meter/std": 0.0723068043589592, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.970588207244873, "rewards/repeat_penalty/std": 0.03144249692559242, "rewards/total_composite/mean": 0.8223901987075806, "rewards/total_composite/std": 0.05969366803765297, "reward": 0.8223901987075806, "reward_std": 0.05969364941120148, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04698567092418671, "sampling/sampling_logp_difference/max": 1.767293930053711, "sampling/importance_sampling_ratio/min": 0.1707945466041565, "sampling/importance_sampling_ratio/mean": 1.0036664009094238, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3467197176069021, "clip_ratio/low_mean": 0.0235387550201267, "clip_ratio/low_min": 0.0235387550201267, "clip_ratio/high_mean": 0.012042732443660498, "clip_ratio/high_max": 0.012042732443660498, "clip_ratio/region_mean": 0.0355814874637872, "reward_total_mean": 0.8223901987075806, "reward_meter_mean": 0.9690572023391724, "reward_meter_std": 0.0723068043589592, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.970588207244873, "reward_repeat_penalty_std": 0.03144249692559242, "reward_total_composite_mean": 0.8223901987075806, "reward_total_composite_std": 0.05969366803765297} {"timestamp_utc": "2026-04-12T02:21:52Z", "mode": "train", "global_step": 2596, "epoch": 0.10426959071374062, "loss": 0.112, "grad_norm": 1.1440000534057617, "learning_rate": 2.1363636363636365e-06, "num_tokens": 5882790.0, "completions/mean_length": 501.125, "completions/min_length": 470.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 499.5714416503906, "completions/min_terminated_length": 470.0, "completions/max_terminated_length": 510.0, "rewards/meter/mean": 0.9981234073638916, "rewards/meter/std": 0.0008658914593979716, "rewards/count_adherence/mean": 0.8046875, "rewards/count_adherence/std": 0.022097086533904076, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9128260612487793, "rewards/repeat_penalty/std": 0.08217112720012665, "rewards/total_composite/mean": 0.7341695427894592, "rewards/total_composite/std": 0.07871917635202408, "reward": 0.7341695427894592, "reward_std": 0.07871916890144348, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.048293329775333405, "sampling/sampling_logp_difference/max": 1.6189045906066895, "sampling/importance_sampling_ratio/min": 0.19811558723449707, "sampling/importance_sampling_ratio/mean": 1.011509895324707, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.34851858019828796, "clip_ratio/low_mean": 0.005782836233265698, "clip_ratio/low_min": 0.005782836233265698, "clip_ratio/high_mean": 0.01985564432106912, "clip_ratio/high_max": 0.01985564432106912, "clip_ratio/region_mean": 0.02563848055433482, "reward_total_mean": 0.7341695427894592, "reward_meter_mean": 0.9981234073638916, "reward_meter_std": 0.0008658914593979716, "reward_count_adherence_mean": 0.8046875, "reward_count_adherence_std": 0.022097086533904076, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9128260612487793, "reward_repeat_penalty_std": 0.08217112720012665, "reward_total_composite_mean": 0.7341695427894592, "reward_total_composite_std": 0.07871917635202408} {"timestamp_utc": "2026-04-12T02:21:57Z", "mode": "train", "global_step": 2597, "epoch": 0.10430975619552557, "loss": -0.0172, "grad_norm": 3.617664098739624, "learning_rate": 2.133333333333334e-06, "num_tokens": 5884995.0, "completions/mean_length": 122.625, "completions/min_length": 113.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.625, "completions/min_terminated_length": 113.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.9898524284362793, "rewards/meter/std": 0.005112205166369677, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8928571939468384, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.8836444616317749, "rewards/total_composite/std": 0.06303286552429199, "reward": 0.8836444616317749, "reward_std": 0.0630328580737114, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.034280113875865936, "sampling/sampling_logp_difference/max": 1.1297893524169922, "sampling/importance_sampling_ratio/min": 0.32310131192207336, "sampling/importance_sampling_ratio/mean": 1.013502597808838, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3435132773593068, "clip_ratio/low_mean": 0.01861410157289356, "clip_ratio/low_min": 0.01861410157289356, "clip_ratio/high_mean": 0.010750473011285067, "clip_ratio/high_max": 0.010750473011285067, "clip_ratio/region_mean": 0.029364574584178627, "reward_total_mean": 0.8836444616317749, "reward_meter_mean": 0.9898524284362793, "reward_meter_std": 0.005112205166369677, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8928571939468384, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.8836444616317749, "reward_total_composite_std": 0.06303286552429199} {"timestamp_utc": "2026-04-12T02:22:01Z", "mode": "train", "global_step": 2598, "epoch": 0.10434992167731053, "loss": -0.0118, "grad_norm": 7.509369850158691, "learning_rate": 2.1303030303030306e-06, "num_tokens": 5886546.0, "completions/mean_length": 28.875, "completions/min_length": 28.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.875, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9953750371932983, "rewards/meter/std": 0.0010425866348668933, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9953750371932983, "rewards/total_composite/std": 0.0010425866348668933, "reward": 0.9953750371932983, "reward_std": 0.0010425925720483065, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00547828758135438, "sampling/sampling_logp_difference/max": 1.0384702682495117, "sampling/importance_sampling_ratio/min": 0.35399580001831055, "sampling/importance_sampling_ratio/mean": 0.9976253509521484, "sampling/importance_sampling_ratio/max": 1.1243258714675903, "entropy": 0.006841129012173042, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9953750371932983, "reward_meter_mean": 0.9953750371932983, "reward_meter_std": 0.0010425866348668933, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9953750371932983, "reward_total_composite_std": 0.0010425866348668933} {"timestamp_utc": "2026-04-12T02:22:06Z", "mode": "train", "global_step": 2599, "epoch": 0.10439008715909548, "loss": 0.0041, "grad_norm": 1.446718692779541, "learning_rate": 2.1272727272727275e-06, "num_tokens": 5888682.0, "completions/mean_length": 98.0, "completions/min_length": 97.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.0, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.9992596507072449, "rewards/meter/std": 0.0001470481656724587, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992596507072449, "rewards/total_composite/std": 0.0001470481656724587, "reward": 0.9992596507072449, "reward_std": 0.000147052516695112, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011022298596799374, "sampling/sampling_logp_difference/max": 0.9700416326522827, "sampling/importance_sampling_ratio/min": 0.3790672719478607, "sampling/importance_sampling_ratio/mean": 0.9985703229904175, "sampling/importance_sampling_ratio/max": 1.4146671295166016, "entropy": 0.06834368547424674, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.007614409434609115, "clip_ratio/high_max": 0.007614409434609115, "clip_ratio/region_mean": 0.007614409434609115, "reward_total_mean": 0.9992596507072449, "reward_meter_mean": 0.9992596507072449, "reward_meter_std": 0.0001470481656724587, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992596507072449, "reward_total_composite_std": 0.0001470481656724587} {"timestamp_utc": "2026-04-12T02:22:10Z", "mode": "train", "global_step": 2600, "epoch": 0.10443025264088043, "loss": 0.0094, "grad_norm": 5.452374458312988, "learning_rate": 2.1242424242424243e-06, "num_tokens": 5890094.0, "completions/mean_length": 34.5, "completions/min_length": 34.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.5, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9899568557739258, "rewards/meter/std": 0.008095722645521164, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9899568557739258, "rewards/total_composite/std": 0.008095722645521164, "reward": 0.9899568557739258, "reward_std": 0.00809571985155344, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02594204805791378, "sampling/sampling_logp_difference/max": 1.6147489547729492, "sampling/importance_sampling_ratio/min": 0.19894060492515564, "sampling/importance_sampling_ratio/mean": 0.9940556287765503, "sampling/importance_sampling_ratio/max": 1.2015459537506104, "entropy": 0.1272981008514762, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.007352941203862429, "clip_ratio/high_max": 0.007352941203862429, "clip_ratio/region_mean": 0.007352941203862429, "reward_total_mean": 0.9899568557739258, "reward_meter_mean": 0.9899568557739258, "reward_meter_std": 0.008095722645521164, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9899568557739258, "reward_total_composite_std": 0.008095722645521164} {"timestamp_utc": "2026-04-12T02:23:28Z", "mode": "eval", "global_step": 2600, "epoch": 0.10443025264088043, "eval_loss": NaN, "eval_runtime": 77.3652, "eval_samples_per_second": 1.344, "eval_steps_per_second": 0.168, "eval_num_tokens": 5890094.0, "eval_completions/mean_length": 209.5096153846154, "eval_completions/min_length": 60.92307692307692, "eval_completions/max_length": 414.3076923076923, "eval_completions/clipped_ratio": 0.019230769230769232, "eval_completions/mean_terminated_length": 203.01236314039963, "eval_completions/min_terminated_length": 60.92307692307692, "eval_completions/max_terminated_length": 392.0, "eval_rewards/meter/mean": 0.7689516819440402, "eval_rewards/meter/std": 0.3567994758486748, "eval_rewards/count_adherence/mean": 0.9526959611819341, "eval_rewards/count_adherence/std": 0.0694849301989262, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/repeat_penalty/mean": 0.92968055835137, "eval_rewards/repeat_penalty/std": 0.09778400797110337, "eval_rewards/total_composite/mean": 0.6867790864064143, "eval_rewards/total_composite/std": 0.3467414522400269, "eval_reward": 0.6867790864064143, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.03625713331768146, "eval_sampling/sampling_logp_difference/max": 1.1111029111422026, "eval_sampling/importance_sampling_ratio/min": 0.3349975932102937, "eval_sampling/importance_sampling_ratio/mean": 1.0097874861497145, "eval_sampling/importance_sampling_ratio/max": 1.5790529526196992, "eval_entropy": 0.40257097207582915, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6867790864064143, "eval_reward_meter_mean": 0.7689516819440402, "eval_reward_meter_std": 0.3567994758486748, "eval_reward_count_adherence_mean": 0.9526959611819341, "eval_reward_count_adherence_std": 0.0694849301989262, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_repeat_penalty_mean": 0.92968055835137, "eval_reward_repeat_penalty_std": 0.09778400797110337, "eval_reward_total_composite_mean": 0.6867790864064143, "eval_reward_total_composite_std": 0.3467414522400269} {"timestamp_utc": "2026-04-12T02:23:35Z", "mode": "train", "global_step": 2601, "epoch": 0.10447041812266539, "loss": -0.0003, "grad_norm": 1.6252681016921997, "learning_rate": 2.1212121212121216e-06, "num_tokens": 5891951.0, "completions/mean_length": 72.125, "completions/min_length": 72.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.125, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9991148710250854, "rewards/meter/std": 0.00010878306056838483, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991148710250854, "rewards/total_composite/std": 0.00010878306056838483, "reward": 0.9991148710250854, "reward_std": 0.00010877339082071558, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010965629480779171, "sampling/sampling_logp_difference/max": 1.061844825744629, "sampling/importance_sampling_ratio/min": 0.3458172678947449, "sampling/importance_sampling_ratio/mean": 1.0020719766616821, "sampling/importance_sampling_ratio/max": 1.4661310911178589, "entropy": 0.06924279127269983, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0017123287543654442, "clip_ratio/high_max": 0.0017123287543654442, "clip_ratio/region_mean": 0.0017123287543654442, "reward_total_mean": 0.9991148710250854, "reward_meter_mean": 0.9991148710250854, "reward_meter_std": 0.00010878306056838483, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9991148710250854, "reward_total_composite_std": 0.00010878306056838483} {"timestamp_utc": "2026-04-12T02:23:40Z", "mode": "train", "global_step": 2602, "epoch": 0.10451058360445034, "loss": -0.0044, "grad_norm": 8.062291145324707, "learning_rate": 2.1181818181818184e-06, "num_tokens": 5893679.0, "completions/mean_length": 68.0, "completions/min_length": 65.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.0, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9672091007232666, "rewards/meter/std": 0.05921924114227295, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9672091007232666, "rewards/total_composite/std": 0.05921924114227295, "reward": 0.9672091007232666, "reward_std": 0.059219252318143845, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03210901468992233, "sampling/sampling_logp_difference/max": 0.8111929893493652, "sampling/importance_sampling_ratio/min": 0.444327712059021, "sampling/importance_sampling_ratio/mean": 1.0095655918121338, "sampling/importance_sampling_ratio/max": 1.7304493188858032, "entropy": 0.24536396749317646, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/high_mean": 0.020244444953277707, "clip_ratio/high_max": 0.020244444953277707, "clip_ratio/region_mean": 0.022138384403660893, "reward_total_mean": 0.9672091007232666, "reward_meter_mean": 0.9672091007232666, "reward_meter_std": 0.05921924114227295, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9672091007232666, "reward_total_composite_std": 0.05921924114227295} {"timestamp_utc": "2026-04-12T02:23:45Z", "mode": "train", "global_step": 2603, "epoch": 0.1045507490862353, "loss": -0.002, "grad_norm": 0.387520432472229, "learning_rate": 2.1151515151515152e-06, "num_tokens": 5895490.0, "completions/mean_length": 72.375, "completions/min_length": 72.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.375, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9991236925125122, "rewards/meter/std": 0.00013291009236127138, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991236925125122, "rewards/total_composite/std": 0.00013291009236127138, "reward": 0.9991236925125122, "reward_std": 0.00013291911454871297, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00656279968097806, "sampling/sampling_logp_difference/max": 0.30100059509277344, "sampling/importance_sampling_ratio/min": 0.8066705465316772, "sampling/importance_sampling_ratio/mean": 1.0055519342422485, "sampling/importance_sampling_ratio/max": 1.3512102365493774, "entropy": 0.05477497586980462, "clip_ratio/low_mean": 0.0034246575087308884, "clip_ratio/low_min": 0.0034246575087308884, "clip_ratio/high_mean": 0.0017123287543654442, "clip_ratio/high_max": 0.0017123287543654442, "clip_ratio/region_mean": 0.0051369862630963326, "reward_total_mean": 0.9991236925125122, "reward_meter_mean": 0.9991236925125122, "reward_meter_std": 0.00013291009236127138, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9991236925125122, "reward_total_composite_std": 0.00013291009236127138} {"timestamp_utc": "2026-04-12T02:23:49Z", "mode": "train", "global_step": 2604, "epoch": 0.10459091456802025, "loss": 0.0218, "grad_norm": 4.805464267730713, "learning_rate": 2.1121212121212125e-06, "num_tokens": 5897338.0, "completions/mean_length": 68.0, "completions/min_length": 67.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9956408739089966, "rewards/meter/std": 0.004533765371888876, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9956408739089966, "rewards/total_composite/std": 0.004533765371888876, "reward": 0.9956408739089966, "reward_std": 0.0045337737537920475, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015114509500563145, "sampling/sampling_logp_difference/max": 1.2838521003723145, "sampling/importance_sampling_ratio/min": 0.2769683301448822, "sampling/importance_sampling_ratio/mean": 1.0015579462051392, "sampling/importance_sampling_ratio/max": 1.9683966636657715, "entropy": 0.07368437433615327, "clip_ratio/low_mean": 0.0034722222480922937, "clip_ratio/low_min": 0.0034722222480922937, "clip_ratio/high_mean": 0.007380377617664635, "clip_ratio/high_max": 0.007380377617664635, "clip_ratio/region_mean": 0.010852599865756929, "reward_total_mean": 0.9956408739089966, "reward_meter_mean": 0.9956408739089966, "reward_meter_std": 0.004533765371888876, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9956408739089966, "reward_total_composite_std": 0.004533765371888876} {"timestamp_utc": "2026-04-12T02:23:54Z", "mode": "train", "global_step": 2605, "epoch": 0.1046310800498052, "loss": -0.0034, "grad_norm": 3.96225905418396, "learning_rate": 2.1090909090909093e-06, "num_tokens": 5899077.0, "completions/mean_length": 59.375, "completions/min_length": 56.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.375, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9955159425735474, "rewards/meter/std": 0.0011379423085600138, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9955159425735474, "rewards/total_composite/std": 0.0011379423085600138, "reward": 0.9955159425735474, "reward_std": 0.001137925311923027, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022009698674082756, "sampling/sampling_logp_difference/max": 0.7918639183044434, "sampling/importance_sampling_ratio/min": 0.45299965143203735, "sampling/importance_sampling_ratio/mean": 1.0065569877624512, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.18665459845215082, "clip_ratio/low_mean": 0.00836864416487515, "clip_ratio/low_min": 0.00836864416487515, "clip_ratio/high_mean": 0.018881317228078842, "clip_ratio/high_max": 0.018881317228078842, "clip_ratio/region_mean": 0.027249961392953992, "reward_total_mean": 0.9955159425735474, "reward_meter_mean": 0.9955159425735474, "reward_meter_std": 0.0011379423085600138, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9955159425735474, "reward_total_composite_std": 0.0011379423085600138} {"timestamp_utc": "2026-04-12T02:23:59Z", "mode": "train", "global_step": 2606, "epoch": 0.10467124553159016, "loss": 0.015, "grad_norm": 3.778212070465088, "learning_rate": 2.106060606060606e-06, "num_tokens": 5901663.0, "completions/mean_length": 137.25, "completions/min_length": 135.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 137.25, "completions/min_terminated_length": 135.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.997505784034729, "rewards/meter/std": 0.002905552973970771, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.997505784034729, "rewards/total_composite/std": 0.002905552973970771, "reward": 0.997505784034729, "reward_std": 0.0029055566992610693, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05034590885043144, "sampling/sampling_logp_difference/max": 1.172276496887207, "sampling/importance_sampling_ratio/min": 0.30966120958328247, "sampling/importance_sampling_ratio/mean": 1.0051594972610474, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.44699474796652794, "clip_ratio/low_mean": 0.010766045656055212, "clip_ratio/low_min": 0.010766045656055212, "clip_ratio/high_mean": 0.04019012441858649, "clip_ratio/high_max": 0.04019012441858649, "clip_ratio/region_mean": 0.050956170074641705, "reward_total_mean": 0.997505784034729, "reward_meter_mean": 0.997505784034729, "reward_meter_std": 0.002905552973970771, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.997505784034729, "reward_total_composite_std": 0.002905552973970771} {"timestamp_utc": "2026-04-12T02:24:05Z", "mode": "train", "global_step": 2607, "epoch": 0.10471141101337511, "loss": 0.0094, "grad_norm": 2.1808922290802, "learning_rate": 2.103030303030303e-06, "num_tokens": 5904846.0, "completions/mean_length": 195.875, "completions/min_length": 190.0, "completions/max_length": 199.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 195.875, "completions/min_terminated_length": 190.0, "completions/max_terminated_length": 199.0, "rewards/meter/mean": 0.9988412261009216, "rewards/meter/std": 0.0004946960834786296, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.9711034893989563, "rewards/total_composite/std": 0.05153718218207359, "reward": 0.9711034893989563, "reward_std": 0.05153718218207359, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04786364734172821, "sampling/sampling_logp_difference/max": 1.6877174377441406, "sampling/importance_sampling_ratio/min": 0.18494117259979248, "sampling/importance_sampling_ratio/mean": 1.0106346607208252, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3686945028603077, "clip_ratio/low_mean": 0.0025254478096030653, "clip_ratio/low_min": 0.0025254478096030653, "clip_ratio/high_mean": 0.023096036864444613, "clip_ratio/high_max": 0.023096036864444613, "clip_ratio/region_mean": 0.02562148467404768, "reward_total_mean": 0.9711034893989563, "reward_meter_mean": 0.9988412261009216, "reward_meter_std": 0.0004946960834786296, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_total_composite_mean": 0.9711034893989563, "reward_total_composite_std": 0.05153718218207359} {"timestamp_utc": "2026-04-12T02:24:11Z", "mode": "train", "global_step": 2608, "epoch": 0.10475157649516006, "loss": -0.0114, "grad_norm": 2.9761064052581787, "learning_rate": 2.1000000000000002e-06, "num_tokens": 5907083.0, "completions/mean_length": 118.625, "completions/min_length": 115.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.625, "completions/min_terminated_length": 115.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.9947597980499268, "rewards/meter/std": 0.010325920768082142, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9947597980499268, "rewards/total_composite/std": 0.010325920768082142, "reward": 0.9947597980499268, "reward_std": 0.010325928218662739, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03754591941833496, "sampling/sampling_logp_difference/max": 1.3222484588623047, "sampling/importance_sampling_ratio/min": 0.26653534173965454, "sampling/importance_sampling_ratio/mean": 1.0031919479370117, "sampling/importance_sampling_ratio/max": 1.8022807836532593, "entropy": 0.283309917896986, "clip_ratio/low_mean": 0.0021739129442721605, "clip_ratio/low_min": 0.0021739129442721605, "clip_ratio/high_mean": 0.020967748598195612, "clip_ratio/high_max": 0.020967748598195612, "clip_ratio/region_mean": 0.023141661542467773, "reward_total_mean": 0.9947597980499268, "reward_meter_mean": 0.9947597980499268, "reward_meter_std": 0.010325920768082142, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9947597980499268, "reward_total_composite_std": 0.010325920768082142} {"timestamp_utc": "2026-04-12T02:24:15Z", "mode": "train", "global_step": 2609, "epoch": 0.10479174197694502, "loss": 0.0134, "grad_norm": 5.4968791007995605, "learning_rate": 2.096969696969697e-06, "num_tokens": 5908961.0, "completions/mean_length": 68.75, "completions/min_length": 66.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.75, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9976909756660461, "rewards/meter/std": 0.003610314568504691, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9976909756660461, "rewards/total_composite/std": 0.003610314568504691, "reward": 0.9976909756660461, "reward_std": 0.00361032597720623, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03931521996855736, "sampling/sampling_logp_difference/max": 1.6529312133789062, "sampling/importance_sampling_ratio/min": 0.19148778915405273, "sampling/importance_sampling_ratio/mean": 1.0021377801895142, "sampling/importance_sampling_ratio/max": 1.6728876829147339, "entropy": 0.3395673893392086, "clip_ratio/low_mean": 0.008980331476777792, "clip_ratio/low_min": 0.008980331476777792, "clip_ratio/high_mean": 0.029262988013215363, "clip_ratio/high_max": 0.029262988013215363, "clip_ratio/region_mean": 0.038243319489993155, "reward_total_mean": 0.9976909756660461, "reward_meter_mean": 0.9976909756660461, "reward_meter_std": 0.003610314568504691, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9976909756660461, "reward_total_composite_std": 0.003610314568504691} {"timestamp_utc": "2026-04-12T02:24:21Z", "mode": "train", "global_step": 2610, "epoch": 0.10483190745872997, "loss": -0.01, "grad_norm": 2.755105495452881, "learning_rate": 2.093939393939394e-06, "num_tokens": 5911873.0, "completions/mean_length": 191.0, "completions/min_length": 170.0, "completions/max_length": 205.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 191.0, "completions/min_terminated_length": 170.0, "completions/max_terminated_length": 205.0, "rewards/meter/mean": 0.9434575438499451, "rewards/meter/std": 0.07025814801454544, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.08625820279121399, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9431818127632141, "rewards/repeat_penalty/std": 0.08328413218259811, "rewards/total_composite/mean": 0.832134485244751, "rewards/total_composite/std": 0.10688581317663193, "reward": 0.832134485244751, "reward_std": 0.10688581317663193, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.057438045740127563, "sampling/sampling_logp_difference/max": 2.1229028701782227, "sampling/importance_sampling_ratio/min": 0.11968369781970978, "sampling/importance_sampling_ratio/mean": 1.007143259048462, "sampling/importance_sampling_ratio/max": 1.9386733770370483, "entropy": 0.49758507683873177, "clip_ratio/low_mean": 0.021181484800763428, "clip_ratio/low_min": 0.021181484800763428, "clip_ratio/high_mean": 0.013105392456054688, "clip_ratio/high_max": 0.013105392456054688, "clip_ratio/region_mean": 0.034286877256818116, "reward_total_mean": 0.832134485244751, "reward_meter_mean": 0.9434575438499451, "reward_meter_std": 0.07025814801454544, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.08625820279121399, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9431818127632141, "reward_repeat_penalty_std": 0.08328413218259811, "reward_total_composite_mean": 0.832134485244751, "reward_total_composite_std": 0.10688581317663193} {"timestamp_utc": "2026-04-12T02:24:28Z", "mode": "train", "global_step": 2611, "epoch": 0.10487207294051493, "loss": 0.005, "grad_norm": 1.116234540939331, "learning_rate": 2.090909090909091e-06, "num_tokens": 5915555.0, "completions/mean_length": 247.25, "completions/min_length": 245.0, "completions/max_length": 249.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 247.25, "completions/min_terminated_length": 245.0, "completions/max_terminated_length": 249.0, "rewards/meter/mean": 0.9987331032752991, "rewards/meter/std": 0.00010533389286138117, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7788461446762085, "rewards/repeat_penalty/std": 0.04929768666625023, "rewards/total_composite/mean": 0.7778571844100952, "rewards/total_composite/std": 0.04919851943850517, "reward": 0.7778571844100952, "reward_std": 0.04919853433966637, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01725967787206173, "sampling/sampling_logp_difference/max": 1.344933032989502, "sampling/importance_sampling_ratio/min": 0.2605571746826172, "sampling/importance_sampling_ratio/mean": 1.0057003498077393, "sampling/importance_sampling_ratio/max": 1.6523418426513672, "entropy": 0.13410179316997528, "clip_ratio/low_mean": 0.007064696401357651, "clip_ratio/low_min": 0.007064696401357651, "clip_ratio/high_mean": 0.00305292836856097, "clip_ratio/high_max": 0.00305292836856097, "clip_ratio/region_mean": 0.01011762476991862, "reward_total_mean": 0.7778571844100952, "reward_meter_mean": 0.9987331032752991, "reward_meter_std": 0.00010533389286138117, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7788461446762085, "reward_repeat_penalty_std": 0.04929768666625023, "reward_total_composite_mean": 0.7778571844100952, "reward_total_composite_std": 0.04919851943850517} {"timestamp_utc": "2026-04-12T02:24:32Z", "mode": "train", "global_step": 2612, "epoch": 0.10491223842229988, "loss": -0.02, "grad_norm": 3.173931121826172, "learning_rate": 2.087878787878788e-06, "num_tokens": 5917441.0, "completions/mean_length": 57.75, "completions/min_length": 56.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.75, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.995871365070343, "rewards/meter/std": 0.0009862068109214306, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.995871365070343, "rewards/total_composite/std": 0.0009862068109214306, "reward": 0.995871365070343, "reward_std": 0.0009862068109214306, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01807764358818531, "sampling/sampling_logp_difference/max": 0.9005880355834961, "sampling/importance_sampling_ratio/min": 0.4063306748867035, "sampling/importance_sampling_ratio/mean": 1.0038785934448242, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08830610243603587, "clip_ratio/low_mean": 0.011121554300189018, "clip_ratio/low_min": 0.011121554300189018, "clip_ratio/high_mean": 0.004204352619126439, "clip_ratio/high_max": 0.004204352619126439, "clip_ratio/region_mean": 0.015325906919315457, "reward_total_mean": 0.995871365070343, "reward_meter_mean": 0.995871365070343, "reward_meter_std": 0.0009862068109214306, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.995871365070343, "reward_total_composite_std": 0.0009862068109214306} {"timestamp_utc": "2026-04-12T02:24:40Z", "mode": "train", "global_step": 2613, "epoch": 0.10495240390408483, "loss": 0.0013, "grad_norm": 2.3518784046173096, "learning_rate": 2.0848484848484852e-06, "num_tokens": 5922023.0, "completions/mean_length": 343.75, "completions/min_length": 326.0, "completions/max_length": 360.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 343.75, "completions/min_terminated_length": 326.0, "completions/max_terminated_length": 360.0, "rewards/meter/mean": 0.990763247013092, "rewards/meter/std": 0.004047461785376072, "rewards/count_adherence/mean": 0.759615421295166, "rewards/count_adherence/std": 0.027196412906050682, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9213085174560547, "rewards/repeat_penalty/std": 0.04907441511750221, "rewards/total_composite/mean": 0.6937090158462524, "rewards/total_composite/std": 0.049949031323194504, "reward": 0.6937090158462524, "reward_std": 0.04994901269674301, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05348372459411621, "sampling/sampling_logp_difference/max": 1.539764404296875, "sampling/importance_sampling_ratio/min": 0.21443161368370056, "sampling/importance_sampling_ratio/mean": 1.0127009153366089, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5114922970533371, "clip_ratio/low_mean": 0.01867919461801648, "clip_ratio/low_min": 0.01867919461801648, "clip_ratio/high_mean": 0.021714599570259452, "clip_ratio/high_max": 0.021714599570259452, "clip_ratio/region_mean": 0.04039379418827593, "reward_total_mean": 0.6937090158462524, "reward_meter_mean": 0.990763247013092, "reward_meter_std": 0.004047461785376072, "reward_count_adherence_mean": 0.759615421295166, "reward_count_adherence_std": 0.027196412906050682, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9213085174560547, "reward_repeat_penalty_std": 0.04907441511750221, "reward_total_composite_mean": 0.6937090158462524, "reward_total_composite_std": 0.049949031323194504} {"timestamp_utc": "2026-04-12T02:24:46Z", "mode": "train", "global_step": 2614, "epoch": 0.10499256938586979, "loss": 0.0036, "grad_norm": 4.099353790283203, "learning_rate": 2.081818181818182e-06, "num_tokens": 5924333.0, "completions/mean_length": 118.75, "completions/min_length": 116.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.75, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.9893277883529663, "rewards/meter/std": 0.016893420368433, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.953779935836792, "rewards/total_composite/std": 0.06425921618938446, "reward": 0.953779935836792, "reward_std": 0.06425920873880386, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028960946947336197, "sampling/sampling_logp_difference/max": 1.3063240051269531, "sampling/importance_sampling_ratio/min": 0.2708137333393097, "sampling/importance_sampling_ratio/mean": 1.0034631490707397, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.20165002904832363, "clip_ratio/low_mean": 0.009516695979982615, "clip_ratio/low_min": 0.009516695979982615, "clip_ratio/high_mean": 0.02407875331118703, "clip_ratio/high_max": 0.02407875331118703, "clip_ratio/region_mean": 0.03359544929116964, "reward_total_mean": 0.953779935836792, "reward_meter_mean": 0.9893277883529663, "reward_meter_std": 0.016893420368433, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.953779935836792, "reward_total_composite_std": 0.06425921618938446} {"timestamp_utc": "2026-04-12T02:24:50Z", "mode": "train", "global_step": 2615, "epoch": 0.10503273486765474, "loss": -0.0063, "grad_norm": 2.5716392993927, "learning_rate": 2.078787878787879e-06, "num_tokens": 5926114.0, "completions/mean_length": 69.625, "completions/min_length": 67.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.625, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9630937576293945, "rewards/meter/std": 0.028052091598510742, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9630937576293945, "rewards/total_composite/std": 0.028052091598510742, "reward": 0.9630937576293945, "reward_std": 0.028052086010575294, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04609367623925209, "sampling/sampling_logp_difference/max": 1.8025195598602295, "sampling/importance_sampling_ratio/min": 0.16488294303417206, "sampling/importance_sampling_ratio/mean": 1.0128580331802368, "sampling/importance_sampling_ratio/max": 1.9397474527359009, "entropy": 0.3094298355281353, "clip_ratio/low_mean": 0.010907472460530698, "clip_ratio/low_min": 0.010907472460530698, "clip_ratio/high_mean": 0.015257316990755498, "clip_ratio/high_max": 0.015257316990755498, "clip_ratio/region_mean": 0.026164789451286197, "reward_total_mean": 0.9630937576293945, "reward_meter_mean": 0.9630937576293945, "reward_meter_std": 0.028052091598510742, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9630937576293945, "reward_total_composite_std": 0.028052091598510742} {"timestamp_utc": "2026-04-12T02:24:56Z", "mode": "train", "global_step": 2616, "epoch": 0.1050729003494397, "loss": -0.0043, "grad_norm": 1.3603968620300293, "learning_rate": 2.075757575757576e-06, "num_tokens": 5928878.0, "completions/mean_length": 159.5, "completions/min_length": 156.0, "completions/max_length": 160.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 159.5, "completions/min_terminated_length": 156.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.9918752908706665, "rewards/meter/std": 0.01707914099097252, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.07120776921510696, "rewards/total_composite/mean": 0.8684781789779663, "rewards/total_composite/std": 0.07916970551013947, "reward": 0.8684781789779663, "reward_std": 0.07916972041130066, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00969278160482645, "sampling/sampling_logp_difference/max": 0.5411806106567383, "sampling/importance_sampling_ratio/min": 0.582060694694519, "sampling/importance_sampling_ratio/mean": 1.0036389827728271, "sampling/importance_sampling_ratio/max": 1.4536888599395752, "entropy": 0.09610292688012123, "clip_ratio/low_mean": 0.0007812500116415322, "clip_ratio/low_min": 0.0007812500116415322, "clip_ratio/high_mean": 0.003906250058207661, "clip_ratio/high_max": 0.003906250058207661, "clip_ratio/region_mean": 0.004687500069849193, "reward_total_mean": 0.8684781789779663, "reward_meter_mean": 0.9918752908706665, "reward_meter_std": 0.01707914099097252, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.07120776921510696, "reward_total_composite_mean": 0.8684781789779663, "reward_total_composite_std": 0.07916970551013947} {"timestamp_utc": "2026-04-12T02:25:01Z", "mode": "train", "global_step": 2617, "epoch": 0.10511306583122465, "loss": -0.0021, "grad_norm": 1.5142834186553955, "learning_rate": 2.072727272727273e-06, "num_tokens": 5931264.0, "completions/mean_length": 127.25, "completions/min_length": 124.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.25, "completions/min_terminated_length": 124.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9976711869239807, "rewards/meter/std": 0.0005347781116142869, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9442130923271179, "rewards/total_composite/std": 0.07360443472862244, "reward": 0.9442130923271179, "reward_std": 0.07360443472862244, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011169607751071453, "sampling/sampling_logp_difference/max": 0.7878756523132324, "sampling/importance_sampling_ratio/min": 0.4548099637031555, "sampling/importance_sampling_ratio/mean": 1.001734733581543, "sampling/importance_sampling_ratio/max": 1.5947275161743164, "entropy": 0.11067105457186699, "clip_ratio/low_mean": 0.002922236453741789, "clip_ratio/low_min": 0.002922236453741789, "clip_ratio/high_mean": 0.00300038093701005, "clip_ratio/high_max": 0.00300038093701005, "clip_ratio/region_mean": 0.005922617390751839, "reward_total_mean": 0.9442130923271179, "reward_meter_mean": 0.9976711869239807, "reward_meter_std": 0.0005347781116142869, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.9442130923271179, "reward_total_composite_std": 0.07360443472862244} {"timestamp_utc": "2026-04-12T02:25:05Z", "mode": "train", "global_step": 2618, "epoch": 0.1051532313130096, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.06969696969697e-06, "num_tokens": 5932856.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00027682288782671094, "sampling/sampling_logp_difference/max": 0.002793335122987628, "sampling/importance_sampling_ratio/min": 1.0, "sampling/importance_sampling_ratio/mean": 1.000277042388916, "sampling/importance_sampling_ratio/max": 1.002797245979309, "entropy": 0.002262885362142697, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:25:10Z", "mode": "train", "global_step": 2619, "epoch": 0.10519339679479456, "loss": -0.01, "grad_norm": 3.8841686248779297, "learning_rate": 2.0666666666666666e-06, "num_tokens": 5935062.0, "completions/mean_length": 90.75, "completions/min_length": 87.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.75, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.9945749640464783, "rewards/meter/std": 0.00539820222184062, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9945749640464783, "rewards/total_composite/std": 0.00539820222184062, "reward": 0.9945749640464783, "reward_std": 0.005398200359195471, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.038317468017339706, "sampling/sampling_logp_difference/max": 2.6888628005981445, "sampling/importance_sampling_ratio/min": 0.06795818358659744, "sampling/importance_sampling_ratio/mean": 1.0023733377456665, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.22051831148564816, "clip_ratio/low_mean": 0.011174242943525314, "clip_ratio/low_min": 0.011174242943525314, "clip_ratio/high_mean": 0.030287093366496265, "clip_ratio/high_max": 0.030287093366496265, "clip_ratio/region_mean": 0.04146133631002158, "reward_total_mean": 0.9945749640464783, "reward_meter_mean": 0.9945749640464783, "reward_meter_std": 0.00539820222184062, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9945749640464783, "reward_total_composite_std": 0.00539820222184062} {"timestamp_utc": "2026-04-12T02:25:18Z", "mode": "train", "global_step": 2620, "epoch": 0.10523356227657951, "loss": 0.0003, "grad_norm": 1.9527921676635742, "learning_rate": 2.063636363636364e-06, "num_tokens": 5939163.0, "completions/mean_length": 298.625, "completions/min_length": 293.0, "completions/max_length": 304.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 298.625, "completions/min_terminated_length": 293.0, "completions/max_terminated_length": 304.0, "rewards/meter/mean": 0.9977849721908569, "rewards/meter/std": 0.0011817796621471643, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9632352590560913, "rewards/repeat_penalty/std": 0.04376610368490219, "rewards/total_composite/mean": 0.9610700011253357, "rewards/total_composite/std": 0.042887065559625626, "reward": 0.9610700011253357, "reward_std": 0.042887069284915924, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0335632786154747, "sampling/sampling_logp_difference/max": 2.3920373916625977, "sampling/importance_sampling_ratio/min": 0.09144318848848343, "sampling/importance_sampling_ratio/mean": 1.0045593976974487, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2537948340177536, "clip_ratio/low_mean": 0.009193609352223575, "clip_ratio/low_min": 0.009193609352223575, "clip_ratio/high_mean": 0.016332621686160564, "clip_ratio/high_max": 0.016332621686160564, "clip_ratio/region_mean": 0.02552623103838414, "reward_total_mean": 0.9610700011253357, "reward_meter_mean": 0.9977849721908569, "reward_meter_std": 0.0011817796621471643, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9632352590560913, "reward_repeat_penalty_std": 0.04376610368490219, "reward_total_composite_mean": 0.9610700011253357, "reward_total_composite_std": 0.042887065559625626} {"timestamp_utc": "2026-04-12T02:25:26Z", "mode": "train", "global_step": 2621, "epoch": 0.10527372775836447, "loss": 0.0049, "grad_norm": 1.2172192335128784, "learning_rate": 2.0606060606060607e-06, "num_tokens": 5943660.0, "completions/mean_length": 349.125, "completions/min_length": 319.0, "completions/max_length": 355.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 349.125, "completions/min_terminated_length": 319.0, "completions/max_terminated_length": 355.0, "rewards/meter/mean": 0.9986586570739746, "rewards/meter/std": 0.0001423863577656448, "rewards/count_adherence/mean": 0.987500011920929, "rewards/count_adherence/std": 0.0353553481400013, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8010836243629456, "rewards/repeat_penalty/std": 0.058759476989507675, "rewards/total_composite/mean": 0.7889921069145203, "rewards/total_composite/std": 0.04864564165472984, "reward": 0.7889921069145203, "reward_std": 0.04864564538002014, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02395707368850708, "sampling/sampling_logp_difference/max": 1.3942632675170898, "sampling/importance_sampling_ratio/min": 0.24801570177078247, "sampling/importance_sampling_ratio/mean": 1.0064207315444946, "sampling/importance_sampling_ratio/max": 1.901794672012329, "entropy": 0.18446156568825245, "clip_ratio/low_mean": 0.006353942735586315, "clip_ratio/low_min": 0.006353942735586315, "clip_ratio/high_mean": 0.004175611422397196, "clip_ratio/high_max": 0.004175611422397196, "clip_ratio/region_mean": 0.010529554157983512, "reward_total_mean": 0.7889921069145203, "reward_meter_mean": 0.9986586570739746, "reward_meter_std": 0.0001423863577656448, "reward_count_adherence_mean": 0.987500011920929, "reward_count_adherence_std": 0.0353553481400013, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8010836243629456, "reward_repeat_penalty_std": 0.058759476989507675, "reward_total_composite_mean": 0.7889921069145203, "reward_total_composite_std": 0.04864564165472984} {"timestamp_utc": "2026-04-12T02:25:31Z", "mode": "train", "global_step": 2622, "epoch": 0.10531389324014942, "loss": -0.0037, "grad_norm": 5.598334789276123, "learning_rate": 2.0575757575757576e-06, "num_tokens": 5945475.0, "completions/mean_length": 59.875, "completions/min_length": 58.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.875, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9960442185401917, "rewards/meter/std": 0.0010061725042760372, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9960442185401917, "rewards/total_composite/std": 0.0010061725042760372, "reward": 0.9960442185401917, "reward_std": 0.0010061829816550016, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.036090608686208725, "sampling/sampling_logp_difference/max": 1.571319580078125, "sampling/importance_sampling_ratio/min": 0.20777083933353424, "sampling/importance_sampling_ratio/mean": 0.9990103244781494, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1663585538044572, "clip_ratio/low_mean": 0.006287686061114073, "clip_ratio/low_min": 0.006287686061114073, "clip_ratio/high_mean": 0.016635986510664225, "clip_ratio/high_max": 0.016635986510664225, "clip_ratio/region_mean": 0.022923672571778297, "reward_total_mean": 0.9960442185401917, "reward_meter_mean": 0.9960442185401917, "reward_meter_std": 0.0010061725042760372, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9960442185401917, "reward_total_composite_std": 0.0010061725042760372} {"timestamp_utc": "2026-04-12T02:25:36Z", "mode": "train", "global_step": 2623, "epoch": 0.10535405872193437, "loss": 0.003, "grad_norm": 3.970113515853882, "learning_rate": 2.054545454545455e-06, "num_tokens": 5947273.0, "completions/mean_length": 71.75, "completions/min_length": 71.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.75, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9978034496307373, "rewards/meter/std": 0.004221748560667038, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978034496307373, "rewards/total_composite/std": 0.004221748560667038, "reward": 0.9978034496307373, "reward_std": 0.004221739247441292, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020419280976057053, "sampling/sampling_logp_difference/max": 0.9410266876220703, "sampling/importance_sampling_ratio/min": 0.39022698998451233, "sampling/importance_sampling_ratio/mean": 1.0029164552688599, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12103518098592758, "clip_ratio/low_mean": 0.0017605633474886417, "clip_ratio/low_min": 0.0017605633474886417, "clip_ratio/high_mean": 0.0052327855955809355, "clip_ratio/high_max": 0.0052327855955809355, "clip_ratio/region_mean": 0.006993348943069577, "reward_total_mean": 0.9978034496307373, "reward_meter_mean": 0.9978034496307373, "reward_meter_std": 0.004221748560667038, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9978034496307373, "reward_total_composite_std": 0.004221748560667038} {"timestamp_utc": "2026-04-12T02:25:40Z", "mode": "train", "global_step": 2624, "epoch": 0.10539422420371933, "loss": 0.0286, "grad_norm": 9.467141151428223, "learning_rate": 2.0515151515151517e-06, "num_tokens": 5948959.0, "completions/mean_length": 64.75, "completions/min_length": 62.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.75, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.017361678183078766, "rewards/meter/std": 0.014113202691078186, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.017361678183078766, "rewards/total_composite/std": 0.014113202691078186, "reward": 0.017361678183078766, "reward_std": 0.014113202691078186, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08726956695318222, "sampling/sampling_logp_difference/max": 2.049853563308716, "sampling/importance_sampling_ratio/min": 0.12875375151634216, "sampling/importance_sampling_ratio/mean": 0.9971011877059937, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3771820291876793, "clip_ratio/low_mean": 0.041920434683561325, "clip_ratio/low_min": 0.041920434683561325, "clip_ratio/high_mean": 0.011576013872399926, "clip_ratio/high_max": 0.011576013872399926, "clip_ratio/region_mean": 0.05349644855596125, "reward_total_mean": 0.017361678183078766, "reward_meter_mean": 0.017361678183078766, "reward_meter_std": 0.014113202691078186, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.017361678183078766, "reward_total_composite_std": 0.014113202691078186} {"timestamp_utc": "2026-04-12T02:25:45Z", "mode": "train", "global_step": 2625, "epoch": 0.10543438968550428, "loss": -0.0006, "grad_norm": 1.3292933702468872, "learning_rate": 2.0484848484848485e-06, "num_tokens": 5950439.0, "completions/mean_length": 37.0, "completions/min_length": 37.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.0, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9995855093002319, "rewards/meter/std": 5.365296237869188e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9995855093002319, "rewards/total_composite/std": 5.365296237869188e-05, "reward": 0.9995855093002319, "reward_std": 5.3658961405744776e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004437229596078396, "sampling/sampling_logp_difference/max": 0.6433219909667969, "sampling/importance_sampling_ratio/min": 0.5255436897277832, "sampling/importance_sampling_ratio/mean": 1.000502109527588, "sampling/importance_sampling_ratio/max": 1.0283234119415283, "entropy": 0.022779989056289196, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9995855093002319, "reward_meter_mean": 0.9995855093002319, "reward_meter_std": 5.365296237869188e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9995855093002319, "reward_total_composite_std": 5.365296237869188e-05} {"timestamp_utc": "2026-04-12T02:25:50Z", "mode": "train", "global_step": 2626, "epoch": 0.10547455516728924, "loss": 0.0077, "grad_norm": 2.536959409713745, "learning_rate": 2.0454545454545457e-06, "num_tokens": 5952538.0, "completions/mean_length": 102.375, "completions/min_length": 100.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.375, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.9648023843765259, "rewards/meter/std": 0.04039134085178375, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9648023843765259, "rewards/total_composite/std": 0.04039134085178375, "reward": 0.9648023843765259, "reward_std": 0.04039134085178375, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03456467390060425, "sampling/sampling_logp_difference/max": 1.0130319595336914, "sampling/importance_sampling_ratio/min": 0.3631163537502289, "sampling/importance_sampling_ratio/mean": 1.0097938776016235, "sampling/importance_sampling_ratio/max": 1.6411556005477905, "entropy": 0.3144999463111162, "clip_ratio/low_mean": 0.008605769136920571, "clip_ratio/low_min": 0.008605769136920571, "clip_ratio/high_mean": 0.009794845129363239, "clip_ratio/high_max": 0.009794845129363239, "clip_ratio/region_mean": 0.01840061426628381, "reward_total_mean": 0.9648023843765259, "reward_meter_mean": 0.9648023843765259, "reward_meter_std": 0.04039134085178375, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9648023843765259, "reward_total_composite_std": 0.04039134085178375} {"timestamp_utc": "2026-04-12T02:25:55Z", "mode": "train", "global_step": 2627, "epoch": 0.10551472064907419, "loss": 0.2115, "grad_norm": 6.214305877685547, "learning_rate": 2.0424242424242426e-06, "num_tokens": 5954150.0, "completions/mean_length": 60.5, "completions/min_length": 36.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.5, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9953294396400452, "rewards/meter/std": 0.010616513900458813, "rewards/count_adherence/mean": 0.25, "rewards/count_adherence/std": 0.4629100561141968, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.24602118134498596, "rewards/total_composite/std": 0.4556134343147278, "reward": 0.24602118134498596, "reward_std": 0.4556134045124054, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03473228961229324, "sampling/sampling_logp_difference/max": 1.8016653060913086, "sampling/importance_sampling_ratio/min": 0.16502384841442108, "sampling/importance_sampling_ratio/mean": 1.0146692991256714, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2536655478179455, "clip_ratio/low_mean": 0.012954836362041533, "clip_ratio/low_min": 0.012954836362041533, "clip_ratio/high_mean": 0.012847222620621324, "clip_ratio/high_max": 0.012847222620621324, "clip_ratio/region_mean": 0.025802058982662857, "reward_total_mean": 0.24602118134498596, "reward_meter_mean": 0.9953294396400452, "reward_meter_std": 0.010616513900458813, "reward_count_adherence_mean": 0.25, "reward_count_adherence_std": 0.4629100561141968, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.24602118134498596, "reward_total_composite_std": 0.4556134343147278} {"timestamp_utc": "2026-04-12T02:26:01Z", "mode": "train", "global_step": 2628, "epoch": 0.10555488613085914, "loss": 0.014, "grad_norm": 3.508739709854126, "learning_rate": 2.03939393939394e-06, "num_tokens": 5957532.0, "completions/mean_length": 239.75, "completions/min_length": 233.0, "completions/max_length": 243.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 239.75, "completions/min_terminated_length": 233.0, "completions/max_terminated_length": 243.0, "rewards/meter/mean": 0.9953933954238892, "rewards/meter/std": 0.00770949712023139, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9953933954238892, "rewards/total_composite/std": 0.00770949712023139, "reward": 0.9953933954238892, "reward_std": 0.007709511090070009, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07987385243177414, "sampling/sampling_logp_difference/max": 3.092388153076172, "sampling/importance_sampling_ratio/min": 0.04539341852068901, "sampling/importance_sampling_ratio/mean": 1.0158778429031372, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7806448265910149, "clip_ratio/low_mean": 0.00927626434713602, "clip_ratio/low_min": 0.00927626434713602, "clip_ratio/high_mean": 0.0423777480609715, "clip_ratio/high_max": 0.0423777480609715, "clip_ratio/region_mean": 0.05165401240810752, "reward_total_mean": 0.9953933954238892, "reward_meter_mean": 0.9953933954238892, "reward_meter_std": 0.00770949712023139, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9953933954238892, "reward_total_composite_std": 0.00770949712023139} {"timestamp_utc": "2026-04-12T02:26:09Z", "mode": "train", "global_step": 2629, "epoch": 0.1055950516126441, "loss": 0.014, "grad_norm": 1.9617708921432495, "learning_rate": 2.0363636363636367e-06, "num_tokens": 5961811.0, "completions/mean_length": 317.875, "completions/min_length": 300.0, "completions/max_length": 337.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 317.875, "completions/min_terminated_length": 300.0, "completions/max_terminated_length": 337.0, "rewards/meter/mean": 0.9979603886604309, "rewards/meter/std": 0.0011684981873258948, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0578637570142746, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9593137502670288, "rewards/repeat_penalty/std": 0.06067804619669914, "rewards/total_composite/mean": 0.927051305770874, "rewards/total_composite/std": 0.07678526639938354, "reward": 0.927051305770874, "reward_std": 0.07678526639938354, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.054092179983854294, "sampling/sampling_logp_difference/max": 1.6163597106933594, "sampling/importance_sampling_ratio/min": 0.19862042367458344, "sampling/importance_sampling_ratio/mean": 1.0117024183273315, "sampling/importance_sampling_ratio/max": 1.913344383239746, "entropy": 0.4582854099571705, "clip_ratio/low_mean": 0.012730259681120515, "clip_ratio/low_min": 0.012730259681120515, "clip_ratio/high_mean": 0.018142351880669594, "clip_ratio/high_max": 0.018142351880669594, "clip_ratio/region_mean": 0.03087261156179011, "reward_total_mean": 0.927051305770874, "reward_meter_mean": 0.9979603886604309, "reward_meter_std": 0.0011684981873258948, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0578637570142746, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9593137502670288, "reward_repeat_penalty_std": 0.06067804619669914, "reward_total_composite_mean": 0.927051305770874, "reward_total_composite_std": 0.07678526639938354} {"timestamp_utc": "2026-04-12T02:26:13Z", "mode": "train", "global_step": 2630, "epoch": 0.10563521709442905, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.0333333333333335e-06, "num_tokens": 5963587.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0030186243820935488, "sampling/sampling_logp_difference/max": 0.3742567002773285, "sampling/importance_sampling_ratio/min": 0.6878003478050232, "sampling/importance_sampling_ratio/mean": 0.9990300536155701, "sampling/importance_sampling_ratio/max": 1.1619328260421753, "entropy": 0.015519737149588764, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:26:19Z", "mode": "train", "global_step": 2631, "epoch": 0.105675382576214, "loss": 0.0062, "grad_norm": 1.6037280559539795, "learning_rate": 2.0303030303030303e-06, "num_tokens": 5966253.0, "completions/mean_length": 143.25, "completions/min_length": 142.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 143.25, "completions/min_terminated_length": 142.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.9989732503890991, "rewards/meter/std": 0.00014036186621524394, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8392857313156128, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.838421106338501, "rewards/total_composite/std": 0.05040955916047096, "reward": 0.838421106338501, "reward_std": 0.050409551709890366, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013544322922825813, "sampling/sampling_logp_difference/max": 1.0127596855163574, "sampling/importance_sampling_ratio/min": 0.36321523785591125, "sampling/importance_sampling_ratio/mean": 1.008257508277893, "sampling/importance_sampling_ratio/max": 1.676928997039795, "entropy": 0.09556546807289124, "clip_ratio/low_mean": 0.0017241379246115685, "clip_ratio/low_min": 0.0017241379246115685, "clip_ratio/high_mean": 0.006118966557551175, "clip_ratio/high_max": 0.006118966557551175, "clip_ratio/region_mean": 0.007843104482162744, "reward_total_mean": 0.838421106338501, "reward_meter_mean": 0.9989732503890991, "reward_meter_std": 0.00014036186621524394, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8392857313156128, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.838421106338501, "reward_total_composite_std": 0.05040955916047096} {"timestamp_utc": "2026-04-12T02:26:23Z", "mode": "train", "global_step": 2632, "epoch": 0.10571554805799896, "loss": -0.007, "grad_norm": 5.14970064163208, "learning_rate": 2.0272727272727276e-06, "num_tokens": 5968023.0, "completions/mean_length": 70.25, "completions/min_length": 69.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.25, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9124512672424316, "rewards/meter/std": 0.0726349726319313, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9124512672424316, "rewards/total_composite/std": 0.0726349726319313, "reward": 0.9124512672424316, "reward_std": 0.0726349726319313, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04469385743141174, "sampling/sampling_logp_difference/max": 1.9476492404937744, "sampling/importance_sampling_ratio/min": 0.14260892570018768, "sampling/importance_sampling_ratio/mean": 1.0102332830429077, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3492947220802307, "clip_ratio/low_mean": 0.023193509317934513, "clip_ratio/low_min": 0.023193509317934513, "clip_ratio/high_mean": 0.015901052742265165, "clip_ratio/high_max": 0.015901052742265165, "clip_ratio/region_mean": 0.03909456206019968, "reward_total_mean": 0.9124512672424316, "reward_meter_mean": 0.9124512672424316, "reward_meter_std": 0.0726349726319313, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9124512672424316, "reward_total_composite_std": 0.0726349726319313} {"timestamp_utc": "2026-04-12T02:26:28Z", "mode": "train", "global_step": 2633, "epoch": 0.10575571353978391, "loss": 0.0122, "grad_norm": 2.2837815284729004, "learning_rate": 2.0242424242424244e-06, "num_tokens": 5970465.0, "completions/mean_length": 136.25, "completions/min_length": 133.0, "completions/max_length": 141.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 136.25, "completions/min_terminated_length": 133.0, "completions/max_terminated_length": 141.0, "rewards/meter/mean": 0.9715265035629272, "rewards/meter/std": 0.011229481548070908, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.910714328289032, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.8844000101089478, "rewards/total_composite/std": 0.06688975542783737, "reward": 0.8844000101089478, "reward_std": 0.06688975542783737, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03914301469922066, "sampling/sampling_logp_difference/max": 1.8302927017211914, "sampling/importance_sampling_ratio/min": 0.16036662459373474, "sampling/importance_sampling_ratio/mean": 1.0096205472946167, "sampling/importance_sampling_ratio/max": 1.7595460414886475, "entropy": 0.3410081285983324, "clip_ratio/low_mean": 0.010007655015215278, "clip_ratio/low_min": 0.010007655015215278, "clip_ratio/high_mean": 0.011128432815894485, "clip_ratio/high_max": 0.011128432815894485, "clip_ratio/region_mean": 0.021136087831109762, "reward_total_mean": 0.8844000101089478, "reward_meter_mean": 0.9715265035629272, "reward_meter_std": 0.011229481548070908, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.910714328289032, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.8844000101089478, "reward_total_composite_std": 0.06688975542783737} {"timestamp_utc": "2026-04-12T02:26:33Z", "mode": "train", "global_step": 2634, "epoch": 0.10579587902156887, "loss": 0.0021, "grad_norm": 3.726565361022949, "learning_rate": 2.0212121212121212e-06, "num_tokens": 5972297.0, "completions/mean_length": 71.0, "completions/min_length": 68.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.0, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9706488251686096, "rewards/meter/std": 0.049138810485601425, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9706488251686096, "rewards/total_composite/std": 0.049138810485601425, "reward": 0.9706488251686096, "reward_std": 0.049138810485601425, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07419008761644363, "sampling/sampling_logp_difference/max": 2.089165449142456, "sampling/importance_sampling_ratio/min": 0.12379039824008942, "sampling/importance_sampling_ratio/mean": 1.0056947469711304, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5871141031384468, "clip_ratio/low_mean": 0.0086805559694767, "clip_ratio/low_min": 0.0086805559694767, "clip_ratio/high_mean": 0.04050721391104162, "clip_ratio/high_max": 0.04050721391104162, "clip_ratio/region_mean": 0.04918776988051832, "reward_total_mean": 0.9706488251686096, "reward_meter_mean": 0.9706488251686096, "reward_meter_std": 0.049138810485601425, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9706488251686096, "reward_total_composite_std": 0.049138810485601425} {"timestamp_utc": "2026-04-12T02:26:38Z", "mode": "train", "global_step": 2635, "epoch": 0.10583604450335382, "loss": 0.0001, "grad_norm": 0.053679000586271286, "learning_rate": 2.0181818181818185e-06, "num_tokens": 5974177.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993990659713745, "rewards/meter/std": 1.0326285746486974e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993990659713745, "rewards/total_composite/std": 1.0326285746486974e-06, "reward": 0.9993990659713745, "reward_std": 1.023619461193448e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.002415498485788703, "sampling/sampling_logp_difference/max": 0.5120803117752075, "sampling/importance_sampling_ratio/min": 0.9006934762001038, "sampling/importance_sampling_ratio/mean": 1.0016577243804932, "sampling/importance_sampling_ratio/max": 1.6687591075897217, "entropy": 0.014239851152524352, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993990659713745, "reward_meter_mean": 0.9993990659713745, "reward_meter_std": 1.0326285746486974e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993990659713745, "reward_total_composite_std": 1.0326285746486974e-06} {"timestamp_utc": "2026-04-12T02:26:42Z", "mode": "train", "global_step": 2636, "epoch": 0.10587620998513878, "loss": -0.0168, "grad_norm": 2.7606544494628906, "learning_rate": 2.0151515151515153e-06, "num_tokens": 5976300.0, "completions/mean_length": 90.375, "completions/min_length": 87.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.375, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.9972773790359497, "rewards/meter/std": 0.0004967614077031612, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9972773790359497, "rewards/total_composite/std": 0.0004967614077031612, "reward": 0.9972773790359497, "reward_std": 0.0004967702552676201, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025844357907772064, "sampling/sampling_logp_difference/max": 1.573763370513916, "sampling/importance_sampling_ratio/min": 0.20726369321346283, "sampling/importance_sampling_ratio/mean": 1.0045125484466553, "sampling/importance_sampling_ratio/max": 1.9276334047317505, "entropy": 0.12905630934983492, "clip_ratio/low_mean": 0.009945004130713642, "clip_ratio/low_min": 0.009945004130713642, "clip_ratio/high_mean": 0.008138859295286238, "clip_ratio/high_max": 0.008138859295286238, "clip_ratio/region_mean": 0.01808386342599988, "reward_total_mean": 0.9972773790359497, "reward_meter_mean": 0.9972773790359497, "reward_meter_std": 0.0004967614077031612, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9972773790359497, "reward_total_composite_std": 0.0004967614077031612} {"timestamp_utc": "2026-04-12T02:26:48Z", "mode": "train", "global_step": 2637, "epoch": 0.10591637546692373, "loss": -0.0064, "grad_norm": 2.6201107501983643, "learning_rate": 2.012121212121212e-06, "num_tokens": 5978451.0, "completions/mean_length": 117.875, "completions/min_length": 113.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 117.875, "completions/min_terminated_length": 113.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.9981694221496582, "rewards/meter/std": 0.0008605656330473721, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981694221496582, "rewards/total_composite/std": 0.0008605656330473721, "reward": 0.9981694221496582, "reward_std": 0.0008605782641097903, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.044327329844236374, "sampling/sampling_logp_difference/max": 1.3086605072021484, "sampling/importance_sampling_ratio/min": 0.27018171548843384, "sampling/importance_sampling_ratio/mean": 1.0066555738449097, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.33568100817501545, "clip_ratio/low_mean": 0.010715189506299794, "clip_ratio/low_min": 0.010715189506299794, "clip_ratio/high_mean": 0.007327559753321111, "clip_ratio/high_max": 0.007327559753321111, "clip_ratio/region_mean": 0.018042749259620905, "reward_total_mean": 0.9981694221496582, "reward_meter_mean": 0.9981694221496582, "reward_meter_std": 0.0008605656330473721, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981694221496582, "reward_total_composite_std": 0.0008605656330473721} {"timestamp_utc": "2026-04-12T02:26:52Z", "mode": "train", "global_step": 2638, "epoch": 0.10595654094870868, "loss": 0.0217, "grad_norm": 6.669706344604492, "learning_rate": 2.009090909090909e-06, "num_tokens": 5979937.0, "completions/mean_length": 35.75, "completions/min_length": 34.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.75, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.9986928701400757, "rewards/meter/std": 0.002228210447356105, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9986928701400757, "rewards/total_composite/std": 0.002228210447356105, "reward": 0.9986928701400757, "reward_std": 0.002228200202807784, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02272917330265045, "sampling/sampling_logp_difference/max": 0.7570033073425293, "sampling/importance_sampling_ratio/min": 0.4690699875354767, "sampling/importance_sampling_ratio/mean": 1.0115097761154175, "sampling/importance_sampling_ratio/max": 1.4079704284667969, "entropy": 0.1990708950906992, "clip_ratio/low_mean": 0.003289473708719015, "clip_ratio/low_min": 0.003289473708719015, "clip_ratio/high_mean": 0.017471515806391835, "clip_ratio/high_max": 0.017471515806391835, "clip_ratio/region_mean": 0.02076098951511085, "reward_total_mean": 0.9986928701400757, "reward_meter_mean": 0.9986928701400757, "reward_meter_std": 0.002228210447356105, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9986928701400757, "reward_total_composite_std": 0.002228210447356105} {"timestamp_utc": "2026-04-12T02:26:57Z", "mode": "train", "global_step": 2639, "epoch": 0.10599670643049364, "loss": -0.0005, "grad_norm": 1.0811501741409302, "learning_rate": 2.0060606060606062e-06, "num_tokens": 5982303.0, "completions/mean_length": 106.75, "completions/min_length": 104.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.75, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.9990361332893372, "rewards/meter/std": 0.00013772756210528314, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990361332893372, "rewards/total_composite/std": 0.00013772756210528314, "reward": 0.9990361332893372, "reward_std": 0.00013771916565019637, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014354676008224487, "sampling/sampling_logp_difference/max": 1.5451383590698242, "sampling/importance_sampling_ratio/min": 0.21328236162662506, "sampling/importance_sampling_ratio/mean": 1.0037332773208618, "sampling/importance_sampling_ratio/max": 1.4171427488327026, "entropy": 0.09720816975459456, "clip_ratio/low_mean": 0.003537735901772976, "clip_ratio/low_min": 0.003537735901772976, "clip_ratio/high_mean": 0.004762336844578385, "clip_ratio/high_max": 0.004762336844578385, "clip_ratio/region_mean": 0.008300072746351361, "reward_total_mean": 0.9990361332893372, "reward_meter_mean": 0.9990361332893372, "reward_meter_std": 0.00013772756210528314, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990361332893372, "reward_total_composite_std": 0.00013772756210528314} {"timestamp_utc": "2026-04-12T02:27:01Z", "mode": "train", "global_step": 2640, "epoch": 0.10603687191227859, "loss": 0.0003, "grad_norm": 1.591088891029358, "learning_rate": 2.0030303030303035e-06, "num_tokens": 5984115.0, "completions/mean_length": 71.5, "completions/min_length": 71.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.5, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9991612434387207, "rewards/meter/std": 0.00017454942280892283, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991612434387207, "rewards/total_composite/std": 0.00017454942280892283, "reward": 0.9991612434387207, "reward_std": 0.00017454590124543756, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008529691025614738, "sampling/sampling_logp_difference/max": 0.8294587135314941, "sampling/importance_sampling_ratio/min": 0.4362854063510895, "sampling/importance_sampling_ratio/mean": 1.001218557357788, "sampling/importance_sampling_ratio/max": 1.096482753753662, "entropy": 0.04928429890424013, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0017361111240461469, "clip_ratio/high_max": 0.0017361111240461469, "clip_ratio/region_mean": 0.0017361111240461469, "reward_total_mean": 0.9991612434387207, "reward_meter_mean": 0.9991612434387207, "reward_meter_std": 0.00017454942280892283, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9991612434387207, "reward_total_composite_std": 0.00017454942280892283} {"timestamp_utc": "2026-04-12T02:27:06Z", "mode": "train", "global_step": 2641, "epoch": 0.10607703739406354, "loss": 0.0002, "grad_norm": 0.12622658908367157, "learning_rate": 2.0000000000000003e-06, "num_tokens": 5985907.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993976354598999, "rewards/meter/std": 3.0407550184463616e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993976354598999, "rewards/total_composite/std": 3.0407550184463616e-06, "reward": 0.9993976354598999, "reward_std": 3.030640073120594e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004023827612400055, "sampling/sampling_logp_difference/max": 0.6670032739639282, "sampling/importance_sampling_ratio/min": 0.5132443308830261, "sampling/importance_sampling_ratio/mean": 0.9996961951255798, "sampling/importance_sampling_ratio/max": 1.6674405336380005, "entropy": 0.01789803896099329, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.00390625, "reward_total_mean": 0.9993976354598999, "reward_meter_mean": 0.9993976354598999, "reward_meter_std": 3.0407550184463616e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993976354598999, "reward_total_composite_std": 3.0407550184463616e-06} {"timestamp_utc": "2026-04-12T02:27:10Z", "mode": "train", "global_step": 2642, "epoch": 0.1061172028758485, "loss": 0.0, "grad_norm": 0.3109838664531708, "learning_rate": 1.996969696969697e-06, "num_tokens": 5987731.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993950724601746, "rewards/meter/std": 5.620834144792752e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993950724601746, "rewards/total_composite/std": 5.620834144792752e-06, "reward": 0.9993950724601746, "reward_std": 5.6233166105812415e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00695427879691124, "sampling/sampling_logp_difference/max": 0.8221545219421387, "sampling/importance_sampling_ratio/min": 0.43948376178741455, "sampling/importance_sampling_ratio/mean": 0.9969460964202881, "sampling/importance_sampling_ratio/max": 1.4890631437301636, "entropy": 0.01494286535307765, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/high_mean": 0.001953125, "clip_ratio/high_max": 0.001953125, "clip_ratio/region_mean": 0.005859375, "reward_total_mean": 0.9993950724601746, "reward_meter_mean": 0.9993950724601746, "reward_meter_std": 5.620834144792752e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993950724601746, "reward_total_composite_std": 5.620834144792752e-06} {"timestamp_utc": "2026-04-12T02:27:14Z", "mode": "train", "global_step": 2643, "epoch": 0.10615736835763345, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.993939393939394e-06, "num_tokens": 5989459.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00024456807295791805, "sampling/sampling_logp_difference/max": 0.003122988622635603, "sampling/importance_sampling_ratio/min": 0.9998295307159424, "sampling/importance_sampling_ratio/mean": 1.0002434253692627, "sampling/importance_sampling_ratio/max": 1.003127932548523, "entropy": 0.0020185734319966286, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:27:19Z", "mode": "train", "global_step": 2644, "epoch": 0.1061975338394184, "loss": -0.0016, "grad_norm": 0.6552127003669739, "learning_rate": 1.9909090909090913e-06, "num_tokens": 5991639.0, "completions/mean_length": 97.5, "completions/min_length": 97.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.5, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9979605674743652, "rewards/meter/std": 0.00010473600559635088, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979605674743652, "rewards/total_composite/std": 0.00010473600559635088, "reward": 0.9979605674743652, "reward_std": 0.00010474038572283462, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011263424530625343, "sampling/sampling_logp_difference/max": 1.1425704956054688, "sampling/importance_sampling_ratio/min": 0.31899797916412354, "sampling/importance_sampling_ratio/mean": 0.9999778270721436, "sampling/importance_sampling_ratio/max": 1.362816333770752, "entropy": 0.07244368549436331, "clip_ratio/low_mean": 0.0012886597542092204, "clip_ratio/low_min": 0.0012886597542092204, "clip_ratio/high_mean": 0.006403850042261183, "clip_ratio/high_max": 0.006403850042261183, "clip_ratio/region_mean": 0.007692509796470404, "reward_total_mean": 0.9979605674743652, "reward_meter_mean": 0.9979605674743652, "reward_meter_std": 0.00010473600559635088, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9979605674743652, "reward_total_composite_std": 0.00010473600559635088} {"timestamp_utc": "2026-04-12T02:27:24Z", "mode": "train", "global_step": 2645, "epoch": 0.10623769932120336, "loss": -0.0, "grad_norm": 0.473066508769989, "learning_rate": 1.987878787878788e-06, "num_tokens": 5993422.0, "completions/mean_length": 66.875, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.875, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981163740158081, "rewards/meter/std": 1.9677801901707426e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981163740158081, "rewards/total_composite/std": 1.9677801901707426e-05, "reward": 0.9981163740158081, "reward_std": 1.96640412468696e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004463972058147192, "sampling/sampling_logp_difference/max": 0.4455909729003906, "sampling/importance_sampling_ratio/min": 0.6404457092285156, "sampling/importance_sampling_ratio/mean": 1.000008225440979, "sampling/importance_sampling_ratio/max": 1.1160743236541748, "entropy": 0.04345296695828438, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0018656715983524919, "clip_ratio/high_max": 0.0018656715983524919, "clip_ratio/region_mean": 0.0018656715983524919, "reward_total_mean": 0.9981163740158081, "reward_meter_mean": 0.9981163740158081, "reward_meter_std": 1.9677801901707426e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981163740158081, "reward_total_composite_std": 1.9677801901707426e-05} {"timestamp_utc": "2026-04-12T02:27:28Z", "mode": "train", "global_step": 2646, "epoch": 0.10627786480298831, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.984848484848485e-06, "num_tokens": 5994830.0, "completions/mean_length": 28.0, "completions/min_length": 28.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.0, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9943599104881287, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9943599104881287, "rewards/total_composite/std": 0.0, "reward": 0.9943599104881287, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002391562593402341, "sampling/sampling_logp_difference/max": 0.0038578114472329617, "sampling/importance_sampling_ratio/min": 0.999919056892395, "sampling/importance_sampling_ratio/mean": 1.0002386569976807, "sampling/importance_sampling_ratio/max": 1.003865361213684, "entropy": 0.0018622480711201206, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9943599104881287, "reward_meter_mean": 0.9943599104881287, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9943599104881287, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:27:34Z", "mode": "train", "global_step": 2647, "epoch": 0.10631803028477327, "loss": 0.0201, "grad_norm": 2.459423303604126, "learning_rate": 1.981818181818182e-06, "num_tokens": 5997841.0, "completions/mean_length": 192.375, "completions/min_length": 185.0, "completions/max_length": 199.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 192.375, "completions/min_terminated_length": 185.0, "completions/max_terminated_length": 199.0, "rewards/meter/mean": 0.9985500574111938, "rewards/meter/std": 0.0007222077692858875, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.9708031415939331, "rewards/total_composite/std": 0.05117257311940193, "reward": 0.9708031415939331, "reward_std": 0.05117259919643402, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04225311800837517, "sampling/sampling_logp_difference/max": 1.1989145278930664, "sampling/importance_sampling_ratio/min": 0.30152130126953125, "sampling/importance_sampling_ratio/mean": 1.0101455450057983, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3521938771009445, "clip_ratio/low_mean": 0.006954187294468284, "clip_ratio/low_min": 0.006954187294468284, "clip_ratio/high_mean": 0.023607419105246663, "clip_ratio/high_max": 0.023607419105246663, "clip_ratio/region_mean": 0.030561606399714947, "reward_total_mean": 0.9708031415939331, "reward_meter_mean": 0.9985500574111938, "reward_meter_std": 0.0007222077692858875, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_total_composite_mean": 0.9708031415939331, "reward_total_composite_std": 0.05117257311940193} {"timestamp_utc": "2026-04-12T02:27:39Z", "mode": "train", "global_step": 2648, "epoch": 0.10635819576655822, "loss": -0.0035, "grad_norm": 2.53367280960083, "learning_rate": 1.978787878787879e-06, "num_tokens": 6000235.0, "completions/mean_length": 131.25, "completions/min_length": 130.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.25, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.9992580413818359, "rewards/meter/std": 5.2671864978037775e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9814149141311646, "rewards/total_composite/std": 0.050484977662563324, "reward": 0.9814149141311646, "reward_std": 0.050484977662563324, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014524316415190697, "sampling/sampling_logp_difference/max": 1.7293694019317627, "sampling/importance_sampling_ratio/min": 0.1773962378501892, "sampling/importance_sampling_ratio/mean": 0.9994350075721741, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08133039530366659, "clip_ratio/low_mean": 0.000961538462433964, "clip_ratio/low_min": 0.000961538462433964, "clip_ratio/high_mean": 0.01428450463572517, "clip_ratio/high_max": 0.01428450463572517, "clip_ratio/region_mean": 0.015246043098159134, "reward_total_mean": 0.9814149141311646, "reward_meter_mean": 0.9992580413818359, "reward_meter_std": 5.2671864978037775e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.9814149141311646, "reward_total_composite_std": 0.050484977662563324} {"timestamp_utc": "2026-04-12T02:27:43Z", "mode": "train", "global_step": 2649, "epoch": 0.10639836124834318, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.975757575757576e-06, "num_tokens": 6002123.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00039579044096171856, "sampling/sampling_logp_difference/max": 0.016826428472995758, "sampling/importance_sampling_ratio/min": 0.9833143353462219, "sampling/importance_sampling_ratio/mean": 1.0002467632293701, "sampling/importance_sampling_ratio/max": 1.0139360427856445, "entropy": 0.004253393795806915, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:27:48Z", "mode": "train", "global_step": 2650, "epoch": 0.10643852673012813, "loss": 0.0008, "grad_norm": 0.25559931993484497, "learning_rate": 1.9727272727272727e-06, "num_tokens": 6003906.0, "completions/mean_length": 66.875, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.875, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981404542922974, "rewards/meter/std": 1.4544022633344866e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981404542922974, "rewards/total_composite/std": 1.4544022633344866e-05, "reward": 0.9981404542922974, "reward_std": 1.4552377251675352e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005564166698604822, "sampling/sampling_logp_difference/max": 0.6460444927215576, "sampling/importance_sampling_ratio/min": 0.5241148471832275, "sampling/importance_sampling_ratio/mean": 0.9998576641082764, "sampling/importance_sampling_ratio/max": 1.3914573192596436, "entropy": 0.036395941860973835, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.007519222097471356, "clip_ratio/high_max": 0.007519222097471356, "clip_ratio/region_mean": 0.007519222097471356, "reward_total_mean": 0.9981404542922974, "reward_meter_mean": 0.9981404542922974, "reward_meter_std": 1.4544022633344866e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981404542922974, "reward_total_composite_std": 1.4544022633344866e-05} {"timestamp_utc": "2026-04-12T02:29:02Z", "mode": "eval", "global_step": 2650, "epoch": 0.10643852673012813, "eval_loss": NaN, "eval_runtime": 73.4017, "eval_samples_per_second": 1.417, "eval_steps_per_second": 0.177, "eval_num_tokens": 6003906.0, "eval_completions/mean_length": 202.45192307692307, "eval_completions/min_length": 61.0, "eval_completions/max_length": 391.2307692307692, "eval_completions/clipped_ratio": 0.009615384615384616, "eval_completions/mean_terminated_length": 199.20054978590744, "eval_completions/min_terminated_length": 61.0, "eval_completions/max_terminated_length": 385.53846153846155, "eval_rewards/meter/mean": 0.8157925880872287, "eval_rewards/meter/std": 0.30426162918313193, "eval_rewards/count_adherence/mean": 0.9371546369332534, "eval_rewards/count_adherence/std": 0.08564070440255679, "eval_rewards/arabic_clean/mean": 0.9807692307692307, "eval_rewards/arabic_clean/std": 0.05439282839114849, "eval_rewards/repeat_penalty/mean": 0.9287464526983408, "eval_rewards/repeat_penalty/std": 0.09337464204201332, "eval_rewards/total_composite/mean": 0.7124902972808251, "eval_rewards/total_composite/std": 0.31389044053279436, "eval_reward": 0.7124902972808251, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.03128652188640375, "eval_sampling/sampling_logp_difference/max": 1.0985317963820238, "eval_sampling/importance_sampling_ratio/min": 0.3371729736144726, "eval_sampling/importance_sampling_ratio/mean": 1.0089522416775043, "eval_sampling/importance_sampling_ratio/max": 1.4974250793457031, "eval_entropy": 0.33797027514531064, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7124902972808251, "eval_reward_meter_mean": 0.8157925880872287, "eval_reward_meter_std": 0.30426162918313193, "eval_reward_count_adherence_mean": 0.9371546369332534, "eval_reward_count_adherence_std": 0.08564070440255679, "eval_reward_arabic_clean_mean": 0.9807692307692307, "eval_reward_arabic_clean_std": 0.05439282839114849, "eval_reward_repeat_penalty_mean": 0.9287464526983408, "eval_reward_repeat_penalty_std": 0.09337464204201332, "eval_reward_total_composite_mean": 0.7124902972808251, "eval_reward_total_composite_std": 0.31389044053279436} {"timestamp_utc": "2026-04-12T02:29:12Z", "mode": "train", "global_step": 2651, "epoch": 0.10647869221191308, "loss": -0.0237, "grad_norm": 2.0326192378997803, "learning_rate": 1.96969696969697e-06, "num_tokens": 6008199.0, "completions/mean_length": 316.625, "completions/min_length": 300.0, "completions/max_length": 341.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 316.625, "completions/min_terminated_length": 300.0, "completions/max_terminated_length": 341.0, "rewards/meter/mean": 0.9982768297195435, "rewards/meter/std": 0.0005552266375161707, "rewards/count_adherence/mean": 0.930555522441864, "rewards/count_adherence/std": 0.05750546231865883, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9622548818588257, "rewards/repeat_penalty/std": 0.05441712588071823, "rewards/total_composite/mean": 0.892192542552948, "rewards/total_composite/std": 0.047268956899642944, "reward": 0.892192542552948, "reward_std": 0.047268956899642944, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04812028631567955, "sampling/sampling_logp_difference/max": 2.127549171447754, "sampling/importance_sampling_ratio/min": 0.11912889778614044, "sampling/importance_sampling_ratio/mean": 1.0096842050552368, "sampling/importance_sampling_ratio/max": 1.9213883876800537, "entropy": 0.3517002984881401, "clip_ratio/low_mean": 0.024150803219527006, "clip_ratio/low_min": 0.024150803219527006, "clip_ratio/high_mean": 0.005865102633833885, "clip_ratio/high_max": 0.005865102633833885, "clip_ratio/region_mean": 0.03001590585336089, "reward_total_mean": 0.892192542552948, "reward_meter_mean": 0.9982768297195435, "reward_meter_std": 0.0005552266375161707, "reward_count_adherence_mean": 0.930555522441864, "reward_count_adherence_std": 0.05750546231865883, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9622548818588257, "reward_repeat_penalty_std": 0.05441712588071823, "reward_total_composite_mean": 0.892192542552948, "reward_total_composite_std": 0.047268956899642944} {"timestamp_utc": "2026-04-12T02:29:17Z", "mode": "train", "global_step": 2652, "epoch": 0.10651885769369804, "loss": -0.004, "grad_norm": 2.323451519012451, "learning_rate": 1.9666666666666668e-06, "num_tokens": 6010904.0, "completions/mean_length": 142.125, "completions/min_length": 140.0, "completions/max_length": 144.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 142.125, "completions/min_terminated_length": 140.0, "completions/max_terminated_length": 144.0, "rewards/meter/mean": 0.9992616176605225, "rewards/meter/std": 9.245655382983387e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.874354362487793, "rewards/total_composite/std": 0.050480589270591736, "reward": 0.874354362487793, "reward_std": 0.050480592995882034, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01351867988705635, "sampling/sampling_logp_difference/max": 1.036015510559082, "sampling/importance_sampling_ratio/min": 0.3548658490180969, "sampling/importance_sampling_ratio/mean": 1.0047122240066528, "sampling/importance_sampling_ratio/max": 1.3792026042938232, "entropy": 0.09845188539475203, "clip_ratio/low_mean": 0.004389182257000357, "clip_ratio/low_min": 0.004389182257000357, "clip_ratio/high_mean": 0.0017361111240461469, "clip_ratio/high_max": 0.0017361111240461469, "clip_ratio/region_mean": 0.006125293381046504, "reward_total_mean": 0.874354362487793, "reward_meter_mean": 0.9992616176605225, "reward_meter_std": 9.245655382983387e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.874354362487793, "reward_total_composite_std": 0.050480589270591736} {"timestamp_utc": "2026-04-12T02:29:26Z", "mode": "train", "global_step": 2653, "epoch": 0.10655902317548299, "loss": -0.0161, "grad_norm": 1.9425232410430908, "learning_rate": 1.9636363636363636e-06, "num_tokens": 6015870.0, "completions/mean_length": 414.75, "completions/min_length": 393.0, "completions/max_length": 425.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 414.75, "completions/min_terminated_length": 393.0, "completions/max_terminated_length": 425.0, "rewards/meter/mean": 0.9989011883735657, "rewards/meter/std": 0.0002041146653937176, "rewards/count_adherence/mean": 0.7767857313156128, "rewards/count_adherence/std": 0.02525380253791809, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9577068090438843, "rewards/repeat_penalty/std": 0.017178824171423912, "rewards/total_composite/mean": 0.7432065010070801, "rewards/total_composite/std": 0.030130181461572647, "reward": 0.7432065010070801, "reward_std": 0.03013017773628235, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04433398321270943, "sampling/sampling_logp_difference/max": 1.5436639785766602, "sampling/importance_sampling_ratio/min": 0.21359705924987793, "sampling/importance_sampling_ratio/mean": 1.0107157230377197, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3583802431821823, "clip_ratio/low_mean": 0.0025445292703807354, "clip_ratio/low_min": 0.0025445292703807354, "clip_ratio/high_mean": 0.02485937112942338, "clip_ratio/high_max": 0.02485937112942338, "clip_ratio/region_mean": 0.027403900399804115, "reward_total_mean": 0.7432065010070801, "reward_meter_mean": 0.9989011883735657, "reward_meter_std": 0.0002041146653937176, "reward_count_adherence_mean": 0.7767857313156128, "reward_count_adherence_std": 0.02525380253791809, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9577068090438843, "reward_repeat_penalty_std": 0.017178824171423912, "reward_total_composite_mean": 0.7432065010070801, "reward_total_composite_std": 0.030130181461572647} {"timestamp_utc": "2026-04-12T02:29:31Z", "mode": "train", "global_step": 2654, "epoch": 0.10659918865726795, "loss": -0.0037, "grad_norm": 1.9117748737335205, "learning_rate": 1.960606060606061e-06, "num_tokens": 6017843.0, "completions/mean_length": 72.625, "completions/min_length": 70.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.625, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9993941783905029, "rewards/meter/std": 0.00015994130808394402, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993941783905029, "rewards/total_composite/std": 0.00015994130808394402, "reward": 0.9993941783905029, "reward_std": 0.0001599277020432055, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01229790784418583, "sampling/sampling_logp_difference/max": 0.7019014358520508, "sampling/importance_sampling_ratio/min": 0.4956419765949249, "sampling/importance_sampling_ratio/mean": 1.0008054971694946, "sampling/importance_sampling_ratio/max": 1.5019214153289795, "entropy": 0.06974478112533689, "clip_ratio/low_mean": 0.0017857142956927419, "clip_ratio/low_min": 0.0017857142956927419, "clip_ratio/high_mean": 0.0017361111240461469, "clip_ratio/high_max": 0.0017361111240461469, "clip_ratio/region_mean": 0.0035218254197388887, "reward_total_mean": 0.9993941783905029, "reward_meter_mean": 0.9993941783905029, "reward_meter_std": 0.00015994130808394402, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993941783905029, "reward_total_composite_std": 0.00015994130808394402} {"timestamp_utc": "2026-04-12T02:29:36Z", "mode": "train", "global_step": 2655, "epoch": 0.1066393541390529, "loss": -0.014, "grad_norm": 9.298754692077637, "learning_rate": 1.9575757575757577e-06, "num_tokens": 6019661.0, "completions/mean_length": 71.25, "completions/min_length": 67.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.25, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9782952070236206, "rewards/meter/std": 0.013127398677170277, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9782952070236206, "rewards/total_composite/std": 0.013127398677170277, "reward": 0.9782952070236206, "reward_std": 0.01312740333378315, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06088637933135033, "sampling/sampling_logp_difference/max": 2.4444236755371094, "sampling/importance_sampling_ratio/min": 0.08677612990140915, "sampling/importance_sampling_ratio/mean": 1.0034326314926147, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4243501238524914, "clip_ratio/low_mean": 0.024847375229001045, "clip_ratio/low_min": 0.024847375229001045, "clip_ratio/high_mean": 0.017125858925282955, "clip_ratio/high_max": 0.017125858925282955, "clip_ratio/region_mean": 0.041973234154284, "reward_total_mean": 0.9782952070236206, "reward_meter_mean": 0.9782952070236206, "reward_meter_std": 0.013127398677170277, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9782952070236206, "reward_total_composite_std": 0.013127398677170277} {"timestamp_utc": "2026-04-12T02:29:40Z", "mode": "train", "global_step": 2656, "epoch": 0.10667951962083785, "loss": 0.0, "grad_norm": 0.48045873641967773, "learning_rate": 1.954545454545455e-06, "num_tokens": 6021436.0, "completions/mean_length": 66.875, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.875, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981427788734436, "rewards/meter/std": 1.307969432673417e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981427788734436, "rewards/total_composite/std": 1.307969432673417e-05, "reward": 0.9981427788734436, "reward_std": 1.3079693417239469e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004913026466965675, "sampling/sampling_logp_difference/max": 0.5399925708770752, "sampling/importance_sampling_ratio/min": 0.5827525854110718, "sampling/importance_sampling_ratio/mean": 1.0033403635025024, "sampling/importance_sampling_ratio/max": 1.6235767602920532, "entropy": 0.029891937039792538, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0018939394503831863, "reward_total_mean": 0.9981427788734436, "reward_meter_mean": 0.9981427788734436, "reward_meter_std": 1.307969432673417e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981427788734436, "reward_total_composite_std": 1.307969432673417e-05} {"timestamp_utc": "2026-04-12T02:29:45Z", "mode": "train", "global_step": 2657, "epoch": 0.10671968510262281, "loss": 0.0029, "grad_norm": 6.030494213104248, "learning_rate": 1.9515151515151518e-06, "num_tokens": 6023229.0, "completions/mean_length": 70.125, "completions/min_length": 67.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.125, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9570232629776001, "rewards/meter/std": 0.08651266247034073, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9570232629776001, "rewards/total_composite/std": 0.08651266247034073, "reward": 0.9570232629776001, "reward_std": 0.08651266247034073, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05514438450336456, "sampling/sampling_logp_difference/max": 1.165654182434082, "sampling/importance_sampling_ratio/min": 0.3117186725139618, "sampling/importance_sampling_ratio/mean": 1.0104464292526245, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.44630854949355125, "clip_ratio/low_mean": 0.0074352502124384046, "clip_ratio/low_min": 0.0074352502124384046, "clip_ratio/high_mean": 0.031580700539052486, "clip_ratio/high_max": 0.031580700539052486, "clip_ratio/region_mean": 0.03901595075149089, "reward_total_mean": 0.9570232629776001, "reward_meter_mean": 0.9570232629776001, "reward_meter_std": 0.08651266247034073, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9570232629776001, "reward_total_composite_std": 0.08651266247034073} {"timestamp_utc": "2026-04-12T02:29:49Z", "mode": "train", "global_step": 2658, "epoch": 0.10675985058440776, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.9484848484848486e-06, "num_tokens": 6025013.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00031924626091495156, "sampling/sampling_logp_difference/max": 0.01690489798784256, "sampling/importance_sampling_ratio/min": 0.9832371473312378, "sampling/importance_sampling_ratio/mean": 1.0001012086868286, "sampling/importance_sampling_ratio/max": 1.0096546411514282, "entropy": 0.0046398533741012216, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:29:53Z", "mode": "train", "global_step": 2659, "epoch": 0.10680001606619272, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.945454545454546e-06, "num_tokens": 6026461.0, "completions/mean_length": 28.0, "completions/min_length": 28.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.0, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9943599104881287, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9943599104881287, "rewards/total_composite/std": 0.0, "reward": 0.9943599104881287, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00031867477810010314, "sampling/sampling_logp_difference/max": 0.00753726065158844, "sampling/importance_sampling_ratio/min": 0.9997876882553101, "sampling/importance_sampling_ratio/mean": 1.0003173351287842, "sampling/importance_sampling_ratio/max": 1.0075657367706299, "entropy": 0.002239607958472334, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9943599104881287, "reward_meter_mean": 0.9943599104881287, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9943599104881287, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:29:58Z", "mode": "train", "global_step": 2660, "epoch": 0.10684018154797767, "loss": 0.0104, "grad_norm": 11.615287780761719, "learning_rate": 1.9424242424242427e-06, "num_tokens": 6028445.0, "completions/mean_length": 65.0, "completions/min_length": 63.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.0, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9398962259292603, "rewards/meter/std": 0.01433228887617588, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9398962259292603, "rewards/total_composite/std": 0.01433228887617588, "reward": 0.9398962259292603, "reward_std": 0.014332284219563007, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0898202583193779, "sampling/sampling_logp_difference/max": 2.166830062866211, "sampling/importance_sampling_ratio/min": 0.22292140126228333, "sampling/importance_sampling_ratio/mean": 1.0091454982757568, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.42190510779619217, "clip_ratio/low_mean": 0.0330266528762877, "clip_ratio/low_min": 0.0330266528762877, "clip_ratio/high_mean": 0.03413416654802859, "clip_ratio/high_max": 0.03413416654802859, "clip_ratio/region_mean": 0.06716081942431629, "reward_total_mean": 0.9398962259292603, "reward_meter_mean": 0.9398962259292603, "reward_meter_std": 0.01433228887617588, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9398962259292603, "reward_total_composite_std": 0.01433228887617588} {"timestamp_utc": "2026-04-12T02:30:03Z", "mode": "train", "global_step": 2661, "epoch": 0.10688034702976262, "loss": -0.0063, "grad_norm": 3.7002193927764893, "learning_rate": 1.9393939393939395e-06, "num_tokens": 6030265.0, "completions/mean_length": 60.5, "completions/min_length": 57.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.5, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9970978498458862, "rewards/meter/std": 0.00044736076961271465, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9970978498458862, "rewards/total_composite/std": 0.00044736076961271465, "reward": 0.9970978498458862, "reward_std": 0.0004473589942790568, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011318504810333252, "sampling/sampling_logp_difference/max": 0.8053483963012695, "sampling/importance_sampling_ratio/min": 0.642192542552948, "sampling/importance_sampling_ratio/mean": 1.0017904043197632, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.053432762157171965, "clip_ratio/low_mean": 0.004385964944958687, "clip_ratio/low_min": 0.004385964944958687, "clip_ratio/high_mean": 0.008196720853447914, "clip_ratio/high_max": 0.008196720853447914, "clip_ratio/region_mean": 0.012582685798406601, "reward_total_mean": 0.9970978498458862, "reward_meter_mean": 0.9970978498458862, "reward_meter_std": 0.00044736076961271465, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9970978498458862, "reward_total_composite_std": 0.00044736076961271465} {"timestamp_utc": "2026-04-12T02:30:07Z", "mode": "train", "global_step": 2662, "epoch": 0.10692051251154758, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.9363636363636363e-06, "num_tokens": 6031961.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002394376788288355, "sampling/sampling_logp_difference/max": 0.002065029926598072, "sampling/importance_sampling_ratio/min": 0.9995288252830505, "sampling/importance_sampling_ratio/mean": 1.0002366304397583, "sampling/importance_sampling_ratio/max": 1.0020672082901, "entropy": 0.0019165669655194506, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:30:12Z", "mode": "train", "global_step": 2663, "epoch": 0.10696067799333253, "loss": -0.0089, "grad_norm": 3.3872175216674805, "learning_rate": 1.9333333333333336e-06, "num_tokens": 6034070.0, "completions/mean_length": 90.625, "completions/min_length": 85.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.625, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.9972575306892395, "rewards/meter/std": 0.0004964250256307423, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9972575306892395, "rewards/total_composite/std": 0.0004964250256307423, "reward": 0.9972575306892395, "reward_std": 0.0004964345716871321, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02105124667286873, "sampling/sampling_logp_difference/max": 1.149244785308838, "sampling/importance_sampling_ratio/min": 0.31687599420547485, "sampling/importance_sampling_ratio/mean": 0.9998160600662231, "sampling/importance_sampling_ratio/max": 1.9014232158660889, "entropy": 0.10070738103240728, "clip_ratio/low_mean": 0.007012551999650896, "clip_ratio/low_min": 0.007012551999650896, "clip_ratio/high_mean": 0.008197940303944051, "clip_ratio/high_max": 0.008197940303944051, "clip_ratio/region_mean": 0.015210492303594947, "reward_total_mean": 0.9972575306892395, "reward_meter_mean": 0.9972575306892395, "reward_meter_std": 0.0004964250256307423, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9972575306892395, "reward_total_composite_std": 0.0004964250256307423} {"timestamp_utc": "2026-04-12T02:30:17Z", "mode": "train", "global_step": 2664, "epoch": 0.10700084347511749, "loss": -0.0067, "grad_norm": 1.5907872915267944, "learning_rate": 1.9303030303030304e-06, "num_tokens": 6035802.0, "completions/mean_length": 70.5, "completions/min_length": 69.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.5, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9993501305580139, "rewards/meter/std": 0.0001369249657727778, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993501305580139, "rewards/total_composite/std": 0.0001369249657727778, "reward": 0.9993501305580139, "reward_std": 0.0001369154779240489, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019528387114405632, "sampling/sampling_logp_difference/max": 0.8889780044555664, "sampling/importance_sampling_ratio/min": 0.4110756516456604, "sampling/importance_sampling_ratio/mean": 1.007887363433838, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10792513377964497, "clip_ratio/low_mean": 0.014288630336523056, "clip_ratio/low_min": 0.014288630336523056, "clip_ratio/high_mean": 0.0017361111240461469, "clip_ratio/high_max": 0.0017361111240461469, "clip_ratio/region_mean": 0.016024741460569203, "reward_total_mean": 0.9993501305580139, "reward_meter_mean": 0.9993501305580139, "reward_meter_std": 0.0001369249657727778, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993501305580139, "reward_total_composite_std": 0.0001369249657727778} {"timestamp_utc": "2026-04-12T02:30:21Z", "mode": "train", "global_step": 2665, "epoch": 0.10704100895690244, "loss": -0.0018, "grad_norm": 12.907752990722656, "learning_rate": 1.9272727272727273e-06, "num_tokens": 6037606.0, "completions/mean_length": 64.5, "completions/min_length": 61.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.5, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.8035573959350586, "rewards/meter/std": 0.2991119921207428, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8035573959350586, "rewards/total_composite/std": 0.2991119921207428, "reward": 0.8035573959350586, "reward_std": 0.2991119921207428, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06501687318086624, "sampling/sampling_logp_difference/max": 1.923792839050293, "sampling/importance_sampling_ratio/min": 0.14605195820331573, "sampling/importance_sampling_ratio/mean": 1.003564715385437, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.25452338345348835, "clip_ratio/low_mean": 0.012102970853447914, "clip_ratio/low_min": 0.012102970853447914, "clip_ratio/high_mean": 0.044199546333402395, "clip_ratio/high_max": 0.044199546333402395, "clip_ratio/region_mean": 0.05630251718685031, "reward_total_mean": 0.8035573959350586, "reward_meter_mean": 0.8035573959350586, "reward_meter_std": 0.2991119921207428, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8035573959350586, "reward_total_composite_std": 0.2991119921207428} {"timestamp_utc": "2026-04-12T02:30:26Z", "mode": "train", "global_step": 2666, "epoch": 0.1070811744386874, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.924242424242424e-06, "num_tokens": 6039238.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00027814743225462735, "sampling/sampling_logp_difference/max": 0.00260642496868968, "sampling/importance_sampling_ratio/min": 0.9999665021896362, "sampling/importance_sampling_ratio/mean": 1.000278115272522, "sampling/importance_sampling_ratio/max": 1.0026098489761353, "entropy": 0.002076218879665248, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:30:31Z", "mode": "train", "global_step": 2667, "epoch": 0.10712133992047235, "loss": 0.0, "grad_norm": 0.08655895292758942, "learning_rate": 1.9212121212121213e-06, "num_tokens": 6041406.0, "completions/mean_length": 98.0, "completions/min_length": 98.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.0, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9993359446525574, "rewards/meter/std": 2.59008470493427e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993359446525574, "rewards/total_composite/std": 2.59008470493427e-06, "reward": 0.9993359446525574, "reward_std": 2.6060056370624807e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0031785282772034407, "sampling/sampling_logp_difference/max": 0.29243287444114685, "sampling/importance_sampling_ratio/min": 0.74644535779953, "sampling/importance_sampling_ratio/mean": 1.0002384185791016, "sampling/importance_sampling_ratio/max": 1.1448330879211426, "entropy": 0.021847927011549473, "clip_ratio/low_mean": 0.0038265305338427424, "clip_ratio/low_min": 0.0038265305338427424, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0038265305338427424, "reward_total_mean": 0.9993359446525574, "reward_meter_mean": 0.9993359446525574, "reward_meter_std": 2.59008470493427e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993359446525574, "reward_total_composite_std": 2.59008470493427e-06} {"timestamp_utc": "2026-04-12T02:30:35Z", "mode": "train", "global_step": 2668, "epoch": 0.1071615054022573, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.918181818181818e-06, "num_tokens": 6043074.0, "completions/mean_length": 36.5, "completions/min_length": 36.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.5, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9996045231819153, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9996045231819153, "rewards/total_composite/std": 0.0, "reward": 0.9996045231819153, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.005902368109673262, "sampling/sampling_logp_difference/max": 0.8077952861785889, "sampling/importance_sampling_ratio/min": 0.44583994150161743, "sampling/importance_sampling_ratio/mean": 0.9983490109443665, "sampling/importance_sampling_ratio/max": 1.073022723197937, "entropy": 0.03415517252869904, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9996045231819153, "reward_meter_mean": 0.9996045231819153, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9996045231819153, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:30:40Z", "mode": "train", "global_step": 2669, "epoch": 0.10720167088404225, "loss": -0.0093, "grad_norm": 5.236100673675537, "learning_rate": 1.9151515151515154e-06, "num_tokens": 6045151.0, "completions/mean_length": 102.625, "completions/min_length": 98.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.625, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.9903435707092285, "rewards/meter/std": 0.007074963301420212, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9903435707092285, "rewards/total_composite/std": 0.007074963301420212, "reward": 0.9903435707092285, "reward_std": 0.00707497913390398, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04934058338403702, "sampling/sampling_logp_difference/max": 1.2319703102111816, "sampling/importance_sampling_ratio/min": 0.2917172312736511, "sampling/importance_sampling_ratio/mean": 1.001228928565979, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39514047652482986, "clip_ratio/low_mean": 0.006191797088831663, "clip_ratio/low_min": 0.006191797088831663, "clip_ratio/high_mean": 0.041350313229486346, "clip_ratio/high_max": 0.041350313229486346, "clip_ratio/region_mean": 0.04754211031831801, "reward_total_mean": 0.9903435707092285, "reward_meter_mean": 0.9903435707092285, "reward_meter_std": 0.007074963301420212, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9903435707092285, "reward_total_composite_std": 0.007074963301420212} {"timestamp_utc": "2026-04-12T02:30:45Z", "mode": "train", "global_step": 2670, "epoch": 0.10724183636582721, "loss": -0.0, "grad_norm": 0.37225642800331116, "learning_rate": 1.9121212121212123e-06, "num_tokens": 6046947.0, "completions/mean_length": 66.5, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.5, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981483817100525, "rewards/meter/std": 1.0958180610032286e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981483817100525, "rewards/total_composite/std": 1.0958180610032286e-05, "reward": 0.9981483817100525, "reward_std": 1.0958180610032286e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006186656653881073, "sampling/sampling_logp_difference/max": 1.6149165630340576, "sampling/importance_sampling_ratio/min": 0.1989072561264038, "sampling/importance_sampling_ratio/mean": 1.0005658864974976, "sampling/importance_sampling_ratio/max": 1.4169889688491821, "entropy": 0.016892601503059268, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0018656715983524919, "clip_ratio/high_max": 0.0018656715983524919, "clip_ratio/region_mean": 0.0018656715983524919, "reward_total_mean": 0.9981483817100525, "reward_meter_mean": 0.9981483817100525, "reward_meter_std": 1.0958180610032286e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981483817100525, "reward_total_composite_std": 1.0958180610032286e-05} {"timestamp_utc": "2026-04-12T02:30:49Z", "mode": "train", "global_step": 2671, "epoch": 0.10728200184761216, "loss": 0.0007, "grad_norm": 7.428125381469727, "learning_rate": 1.9090909090909095e-06, "num_tokens": 6048617.0, "completions/mean_length": 60.75, "completions/min_length": 60.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.75, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9970404505729675, "rewards/meter/std": 0.00046239994117058814, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9970404505729675, "rewards/total_composite/std": 0.00046239994117058814, "reward": 0.9970404505729675, "reward_std": 0.0004624156281352043, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014924735762178898, "sampling/sampling_logp_difference/max": 0.8970005512237549, "sampling/importance_sampling_ratio/min": 0.4077909588813782, "sampling/importance_sampling_ratio/mean": 1.0006794929504395, "sampling/importance_sampling_ratio/max": 1.6209132671356201, "entropy": 0.07036425499245524, "clip_ratio/low_mean": 0.014549180399626493, "clip_ratio/low_min": 0.014549180399626493, "clip_ratio/high_mean": 0.004098360426723957, "clip_ratio/high_max": 0.004098360426723957, "clip_ratio/region_mean": 0.01864754082635045, "reward_total_mean": 0.9970404505729675, "reward_meter_mean": 0.9970404505729675, "reward_meter_std": 0.00046239994117058814, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9970404505729675, "reward_total_composite_std": 0.00046239994117058814} {"timestamp_utc": "2026-04-12T02:30:58Z", "mode": "train", "global_step": 2672, "epoch": 0.10732216732939712, "loss": 0.0098, "grad_norm": 6.031638145446777, "learning_rate": 1.9060606060606064e-06, "num_tokens": 6052766.0, "completions/mean_length": 316.625, "completions/min_length": 309.0, "completions/max_length": 328.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 316.625, "completions/min_terminated_length": 309.0, "completions/max_terminated_length": 328.0, "rewards/meter/mean": 0.9499752521514893, "rewards/meter/std": 0.06226501986384392, "rewards/count_adherence/mean": 0.8409091234207153, "rewards/count_adherence/std": 0.04208274558186531, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.8695175647735596, "rewards/repeat_penalty/std": 0.058811455965042114, "rewards/total_composite/mean": 0.6005241870880127, "rewards/total_composite/std": 0.24652278423309326, "reward": 0.6005241870880127, "reward_std": 0.24652279913425446, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04524172842502594, "sampling/sampling_logp_difference/max": 1.6112018823623657, "sampling/importance_sampling_ratio/min": 0.19964751601219177, "sampling/importance_sampling_ratio/mean": 1.0106778144836426, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.46788543090224266, "clip_ratio/low_mean": 0.0011718750465661287, "clip_ratio/low_min": 0.0011718750465661287, "clip_ratio/high_mean": 0.022553268587216735, "clip_ratio/high_max": 0.022553268587216735, "clip_ratio/region_mean": 0.023725143633782864, "reward_total_mean": 0.6005241870880127, "reward_meter_mean": 0.9499752521514893, "reward_meter_std": 0.06226501986384392, "reward_count_adherence_mean": 0.8409091234207153, "reward_count_adherence_std": 0.04208274558186531, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.8695175647735596, "reward_repeat_penalty_std": 0.058811455965042114, "reward_total_composite_mean": 0.6005241870880127, "reward_total_composite_std": 0.24652278423309326} {"timestamp_utc": "2026-04-12T02:31:02Z", "mode": "train", "global_step": 2673, "epoch": 0.10736233281118207, "loss": -0.0033, "grad_norm": 1.5127347707748413, "learning_rate": 1.9030303030303032e-06, "num_tokens": 6054615.0, "completions/mean_length": 79.125, "completions/min_length": 78.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.125, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.99886155128479, "rewards/meter/std": 0.00016637769294902682, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.99886155128479, "rewards/total_composite/std": 0.00016637769294902682, "reward": 0.99886155128479, "reward_std": 0.00016637658700346947, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0236604493111372, "sampling/sampling_logp_difference/max": 0.8233041763305664, "sampling/importance_sampling_ratio/min": 0.43897879123687744, "sampling/importance_sampling_ratio/mean": 1.0049008131027222, "sampling/importance_sampling_ratio/max": 1.7465615272521973, "entropy": 0.12998810317367315, "clip_ratio/low_mean": 0.00795196380931884, "clip_ratio/low_min": 0.00795196380931884, "clip_ratio/high_mean": 0.006329114083200693, "clip_ratio/high_max": 0.006329114083200693, "clip_ratio/region_mean": 0.014281077892519534, "reward_total_mean": 0.99886155128479, "reward_meter_mean": 0.99886155128479, "reward_meter_std": 0.00016637769294902682, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.99886155128479, "reward_total_composite_std": 0.00016637769294902682} {"timestamp_utc": "2026-04-12T02:31:09Z", "mode": "train", "global_step": 2674, "epoch": 0.10740249829296702, "loss": 0.0073, "grad_norm": 2.1606380939483643, "learning_rate": 1.9000000000000002e-06, "num_tokens": 6058386.0, "completions/mean_length": 268.375, "completions/min_length": 262.0, "completions/max_length": 272.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 268.375, "completions/min_terminated_length": 262.0, "completions/max_terminated_length": 272.0, "rewards/meter/mean": 0.9988313317298889, "rewards/meter/std": 0.0005680599715560675, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.942307710647583, "rewards/repeat_penalty/std": 0.054392825812101364, "rewards/total_composite/mean": 0.9412178993225098, "rewards/total_composite/std": 0.0545286163687706, "reward": 0.9412178993225098, "reward_std": 0.0545286163687706, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.043261099606752396, "sampling/sampling_logp_difference/max": 1.4657955169677734, "sampling/importance_sampling_ratio/min": 0.23089425265789032, "sampling/importance_sampling_ratio/mean": 1.0066032409667969, "sampling/importance_sampling_ratio/max": 1.7729755640029907, "entropy": 0.33767288736999035, "clip_ratio/low_mean": 0.012097747880034149, "clip_ratio/low_min": 0.012097747880034149, "clip_ratio/high_mean": 0.012176374206319451, "clip_ratio/high_max": 0.012176374206319451, "clip_ratio/region_mean": 0.0242741220863536, "reward_total_mean": 0.9412178993225098, "reward_meter_mean": 0.9988313317298889, "reward_meter_std": 0.0005680599715560675, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.942307710647583, "reward_repeat_penalty_std": 0.054392825812101364, "reward_total_composite_mean": 0.9412178993225098, "reward_total_composite_std": 0.0545286163687706} {"timestamp_utc": "2026-04-12T02:31:18Z", "mode": "train", "global_step": 2675, "epoch": 0.10744266377475198, "loss": 0.0064, "grad_norm": 1.5513638257980347, "learning_rate": 1.896969696969697e-06, "num_tokens": 6062752.0, "completions/mean_length": 340.75, "completions/min_length": 337.0, "completions/max_length": 346.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 340.75, "completions/min_terminated_length": 337.0, "completions/max_terminated_length": 346.0, "rewards/meter/mean": 0.9985557794570923, "rewards/meter/std": 0.0006976545555517077, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9117647409439087, "rewards/repeat_penalty/std": 0.05446000397205353, "rewards/total_composite/mean": 0.8193838000297546, "rewards/total_composite/std": 0.04855939745903015, "reward": 0.8193838000297546, "reward_std": 0.04855942353606224, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.040726643055677414, "sampling/sampling_logp_difference/max": 1.2356281280517578, "sampling/importance_sampling_ratio/min": 0.2906521260738373, "sampling/importance_sampling_ratio/mean": 1.0090699195861816, "sampling/importance_sampling_ratio/max": 1.8999286890029907, "entropy": 0.3350350670516491, "clip_ratio/low_mean": 0.009543466847389936, "clip_ratio/low_min": 0.009543466847389936, "clip_ratio/high_mean": 0.014709851937368512, "clip_ratio/high_max": 0.014709851937368512, "clip_ratio/region_mean": 0.02425331878475845, "reward_total_mean": 0.8193838000297546, "reward_meter_mean": 0.9985557794570923, "reward_meter_std": 0.0006976545555517077, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9117647409439087, "reward_repeat_penalty_std": 0.05446000397205353, "reward_total_composite_mean": 0.8193838000297546, "reward_total_composite_std": 0.04855939745903015} {"timestamp_utc": "2026-04-12T02:31:22Z", "mode": "train", "global_step": 2676, "epoch": 0.10748282925653693, "loss": -0.0079, "grad_norm": 3.5680060386657715, "learning_rate": 1.8939393939393941e-06, "num_tokens": 6064157.0, "completions/mean_length": 34.625, "completions/min_length": 34.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.625, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9994817972183228, "rewards/meter/std": 0.0001635835214983672, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994817972183228, "rewards/total_composite/std": 0.0001635835214983672, "reward": 0.9994817972183228, "reward_std": 0.00016358528228010982, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015038218349218369, "sampling/sampling_logp_difference/max": 0.7005414962768555, "sampling/importance_sampling_ratio/min": 0.49631646275520325, "sampling/importance_sampling_ratio/mean": 1.0051720142364502, "sampling/importance_sampling_ratio/max": 1.5120035409927368, "entropy": 0.07462144270539284, "clip_ratio/low_mean": 0.0036764706019312143, "clip_ratio/low_min": 0.0036764706019312143, "clip_ratio/high_mean": 0.0071428571827709675, "clip_ratio/high_max": 0.0071428571827709675, "clip_ratio/region_mean": 0.010819327784702182, "reward_total_mean": 0.9994817972183228, "reward_meter_mean": 0.9994817972183228, "reward_meter_std": 0.0001635835214983672, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994817972183228, "reward_total_composite_std": 0.0001635835214983672} {"timestamp_utc": "2026-04-12T02:31:28Z", "mode": "train", "global_step": 2677, "epoch": 0.10752299473832189, "loss": 0.0237, "grad_norm": 3.107592821121216, "learning_rate": 1.890909090909091e-06, "num_tokens": 6067073.0, "completions/mean_length": 175.5, "completions/min_length": 167.0, "completions/max_length": 184.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 175.5, "completions/min_terminated_length": 167.0, "completions/max_terminated_length": 184.0, "rewards/meter/mean": 0.9970433712005615, "rewards/meter/std": 0.0006079506129026413, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9772727489471436, "rewards/repeat_penalty/std": 0.04208271950483322, "rewards/total_composite/mean": 0.974378228187561, "rewards/total_composite/std": 0.041835807263851166, "reward": 0.974378228187561, "reward_std": 0.041835810989141464, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.033983275294303894, "sampling/sampling_logp_difference/max": 2.001898765563965, "sampling/importance_sampling_ratio/min": 0.135078564286232, "sampling/importance_sampling_ratio/mean": 1.0057967901229858, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.19678651168942451, "clip_ratio/low_mean": 0.007593599148094654, "clip_ratio/low_min": 0.007593599148094654, "clip_ratio/high_mean": 0.029550255741924047, "clip_ratio/high_max": 0.029550255741924047, "clip_ratio/region_mean": 0.0371438548900187, "reward_total_mean": 0.974378228187561, "reward_meter_mean": 0.9970433712005615, "reward_meter_std": 0.0006079506129026413, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9772727489471436, "reward_repeat_penalty_std": 0.04208271950483322, "reward_total_composite_mean": 0.974378228187561, "reward_total_composite_std": 0.041835807263851166} {"timestamp_utc": "2026-04-12T02:31:33Z", "mode": "train", "global_step": 2678, "epoch": 0.10756316022010684, "loss": -0.0038, "grad_norm": 2.5906176567077637, "learning_rate": 1.887878787878788e-06, "num_tokens": 6069055.0, "completions/mean_length": 91.75, "completions/min_length": 88.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.75, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.9974672198295593, "rewards/meter/std": 0.0002358894416829571, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9974672198295593, "rewards/total_composite/std": 0.0002358894416829571, "reward": 0.9974672198295593, "reward_std": 0.00023587615578435361, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025346064940094948, "sampling/sampling_logp_difference/max": 1.5407707691192627, "sampling/importance_sampling_ratio/min": 0.2142159342765808, "sampling/importance_sampling_ratio/mean": 1.0014500617980957, "sampling/importance_sampling_ratio/max": 1.6885485649108887, "entropy": 0.12407996132969856, "clip_ratio/low_mean": 0.004108626628294587, "clip_ratio/low_min": 0.004108626628294587, "clip_ratio/high_mean": 0.014815408736467361, "clip_ratio/high_max": 0.014815408736467361, "clip_ratio/region_mean": 0.01892403536476195, "reward_total_mean": 0.9974672198295593, "reward_meter_mean": 0.9974672198295593, "reward_meter_std": 0.0002358894416829571, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9974672198295593, "reward_total_composite_std": 0.0002358894416829571} {"timestamp_utc": "2026-04-12T02:31:38Z", "mode": "train", "global_step": 2679, "epoch": 0.1076033257018918, "loss": 0.0065, "grad_norm": 3.563011646270752, "learning_rate": 1.884848484848485e-06, "num_tokens": 6070908.0, "completions/mean_length": 67.625, "completions/min_length": 64.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.625, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9295022487640381, "rewards/meter/std": 0.1213918924331665, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9295022487640381, "rewards/total_composite/std": 0.1213918924331665, "reward": 0.9295022487640381, "reward_std": 0.12139188498258591, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.034729912877082825, "sampling/sampling_logp_difference/max": 1.1639995574951172, "sampling/importance_sampling_ratio/min": 0.31223487854003906, "sampling/importance_sampling_ratio/mean": 1.0032422542572021, "sampling/importance_sampling_ratio/max": 1.9054324626922607, "entropy": 0.2862533424049616, "clip_ratio/low_mean": 0.007462686393409967, "clip_ratio/low_min": 0.007462686393409967, "clip_ratio/high_mean": 0.021945541491732, "clip_ratio/high_max": 0.021945541491732, "clip_ratio/region_mean": 0.02940822788514197, "reward_total_mean": 0.9295022487640381, "reward_meter_mean": 0.9295022487640381, "reward_meter_std": 0.1213918924331665, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9295022487640381, "reward_total_composite_std": 0.1213918924331665} {"timestamp_utc": "2026-04-12T02:31:43Z", "mode": "train", "global_step": 2680, "epoch": 0.10764349118367675, "loss": -0.0195, "grad_norm": 8.263280868530273, "learning_rate": 1.8818181818181819e-06, "num_tokens": 6072994.0, "completions/mean_length": 89.75, "completions/min_length": 83.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.75, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.9924962520599365, "rewards/meter/std": 0.013577880337834358, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9924962520599365, "rewards/total_composite/std": 0.013577880337834358, "reward": 0.9924962520599365, "reward_std": 0.01357787661254406, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030782943591475487, "sampling/sampling_logp_difference/max": 1.3811171054840088, "sampling/importance_sampling_ratio/min": 0.2512976825237274, "sampling/importance_sampling_ratio/mean": 0.9978711605072021, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12778305634856224, "clip_ratio/low_mean": 0.0015060240402817726, "clip_ratio/low_min": 0.0015060240402817726, "clip_ratio/high_mean": 0.024966524448245764, "clip_ratio/high_max": 0.024966524448245764, "clip_ratio/region_mean": 0.026472548488527536, "reward_total_mean": 0.9924962520599365, "reward_meter_mean": 0.9924962520599365, "reward_meter_std": 0.013577880337834358, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9924962520599365, "reward_total_composite_std": 0.013577880337834358} {"timestamp_utc": "2026-04-12T02:31:48Z", "mode": "train", "global_step": 2681, "epoch": 0.1076836566654617, "loss": -0.0005, "grad_norm": 2.0896339416503906, "learning_rate": 1.878787878787879e-06, "num_tokens": 6075627.0, "completions/mean_length": 132.125, "completions/min_length": 131.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.125, "completions/min_terminated_length": 131.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.9991657733917236, "rewards/meter/std": 0.0002873947087209672, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991657733917236, "rewards/total_composite/std": 0.0002873947087209672, "reward": 0.9991657733917236, "reward_std": 0.00028738551191054285, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015285594388842583, "sampling/sampling_logp_difference/max": 1.4336557388305664, "sampling/importance_sampling_ratio/min": 0.23843567073345184, "sampling/importance_sampling_ratio/mean": 0.9993587136268616, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.058492994867265224, "clip_ratio/low_mean": 0.006650691735558212, "clip_ratio/low_min": 0.006650691735558212, "clip_ratio/high_mean": 0.004713488393463194, "clip_ratio/high_max": 0.004713488393463194, "clip_ratio/region_mean": 0.011364180129021406, "reward_total_mean": 0.9991657733917236, "reward_meter_mean": 0.9991657733917236, "reward_meter_std": 0.0002873947087209672, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9991657733917236, "reward_total_composite_std": 0.0002873947087209672} {"timestamp_utc": "2026-04-12T02:31:53Z", "mode": "train", "global_step": 2682, "epoch": 0.10772382214724666, "loss": 0.0047, "grad_norm": 2.7295045852661133, "learning_rate": 1.8757575757575757e-06, "num_tokens": 6077792.0, "completions/mean_length": 92.625, "completions/min_length": 91.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.625, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.9971524477005005, "rewards/meter/std": 0.0005293982103466988, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9971524477005005, "rewards/total_composite/std": 0.0005293982103466988, "reward": 0.9971524477005005, "reward_std": 0.0005293957656249404, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03304116055369377, "sampling/sampling_logp_difference/max": 1.4015769958496094, "sampling/importance_sampling_ratio/min": 0.24620838463306427, "sampling/importance_sampling_ratio/mean": 1.0033948421478271, "sampling/importance_sampling_ratio/max": 1.9131104946136475, "entropy": 0.1409629676491022, "clip_ratio/low_mean": 0.021640333347022533, "clip_ratio/low_min": 0.021640333347022533, "clip_ratio/high_mean": 0.010841309442184865, "clip_ratio/high_max": 0.010841309442184865, "clip_ratio/region_mean": 0.0324816427892074, "reward_total_mean": 0.9971524477005005, "reward_meter_mean": 0.9971524477005005, "reward_meter_std": 0.0005293982103466988, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9971524477005005, "reward_total_composite_std": 0.0005293982103466988} {"timestamp_utc": "2026-04-12T02:31:57Z", "mode": "train", "global_step": 2683, "epoch": 0.10776398762903161, "loss": -0.0069, "grad_norm": 13.13068962097168, "learning_rate": 1.872727272727273e-06, "num_tokens": 6079318.0, "completions/mean_length": 42.75, "completions/min_length": 42.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.75, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.953346848487854, "rewards/meter/std": 0.009581836871802807, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.953346848487854, "rewards/total_composite/std": 0.009581836871802807, "reward": 0.953346848487854, "reward_std": 0.009581834077835083, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06227937340736389, "sampling/sampling_logp_difference/max": 1.1528511047363281, "sampling/importance_sampling_ratio/min": 0.3157352805137634, "sampling/importance_sampling_ratio/mean": 1.0015714168548584, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.28468130715191364, "clip_ratio/low_mean": 0.02360188332386315, "clip_ratio/low_min": 0.02360188332386315, "clip_ratio/high_mean": 0.014336680993437767, "clip_ratio/high_max": 0.014336680993437767, "clip_ratio/region_mean": 0.037938564317300916, "reward_total_mean": 0.953346848487854, "reward_meter_mean": 0.953346848487854, "reward_meter_std": 0.009581836871802807, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.953346848487854, "reward_total_composite_std": 0.009581836871802807} {"timestamp_utc": "2026-04-12T02:32:02Z", "mode": "train", "global_step": 2684, "epoch": 0.10780415311081656, "loss": -0.0002, "grad_norm": 0.11564414203166962, "learning_rate": 1.86969696969697e-06, "num_tokens": 6081094.0, "completions/mean_length": 41.0, "completions/min_length": 41.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.0, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.9985654354095459, "rewards/meter/std": 4.657397312257672e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9985654354095459, "rewards/total_composite/std": 4.657397312257672e-06, "reward": 0.9985654354095459, "reward_std": 4.660256763600046e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0033583450131118298, "sampling/sampling_logp_difference/max": 0.5027565956115723, "sampling/importance_sampling_ratio/min": 0.6048609614372253, "sampling/importance_sampling_ratio/mean": 1.0005881786346436, "sampling/importance_sampling_ratio/max": 1.0496618747711182, "entropy": 0.01764180068857968, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985654354095459, "reward_meter_mean": 0.9985654354095459, "reward_meter_std": 4.657397312257672e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9985654354095459, "reward_total_composite_std": 4.657397312257672e-06} {"timestamp_utc": "2026-04-12T02:32:06Z", "mode": "train", "global_step": 2685, "epoch": 0.10784431859260152, "loss": -0.0036, "grad_norm": 5.177779197692871, "learning_rate": 1.8666666666666669e-06, "num_tokens": 6082730.0, "completions/mean_length": 60.5, "completions/min_length": 57.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.5, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9971067905426025, "rewards/meter/std": 0.0004163467965554446, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9971067905426025, "rewards/total_composite/std": 0.0004163467965554446, "reward": 0.9971067905426025, "reward_std": 0.0004163416160736233, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014101680368185043, "sampling/sampling_logp_difference/max": 0.9058688879013062, "sampling/importance_sampling_ratio/min": 0.4041905701160431, "sampling/importance_sampling_ratio/mean": 1.0009123086929321, "sampling/importance_sampling_ratio/max": 1.7392834424972534, "entropy": 0.06518607307225466, "clip_ratio/low_mean": 0.004385964944958687, "clip_ratio/low_min": 0.004385964944958687, "clip_ratio/high_mean": 0.006147540640085936, "clip_ratio/high_max": 0.006147540640085936, "clip_ratio/region_mean": 0.010533505585044622, "reward_total_mean": 0.9971067905426025, "reward_meter_mean": 0.9971067905426025, "reward_meter_std": 0.0004163467965554446, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9971067905426025, "reward_total_composite_std": 0.0004163467965554446} {"timestamp_utc": "2026-04-12T02:32:10Z", "mode": "train", "global_step": 2686, "epoch": 0.10788448407438647, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.863636363636364e-06, "num_tokens": 6084138.0, "completions/mean_length": 32.0, "completions/min_length": 32.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "reward": 0.9992982149124146, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0005363853415474296, "sampling/sampling_logp_difference/max": 0.028025653213262558, "sampling/importance_sampling_ratio/min": 0.9723634123802185, "sampling/importance_sampling_ratio/mean": 0.999911904335022, "sampling/importance_sampling_ratio/max": 1.010701298713684, "entropy": 0.0043548993999138474, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9992982149124146, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:32:20Z", "mode": "train", "global_step": 2687, "epoch": 0.10792464955617143, "loss": -0.0234, "grad_norm": 1.300607681274414, "learning_rate": 1.8606060606060607e-06, "num_tokens": 6090080.0, "completions/mean_length": 472.75, "completions/min_length": 433.0, "completions/max_length": 491.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 472.75, "completions/min_terminated_length": 433.0, "completions/max_terminated_length": 491.0, "rewards/meter/mean": 0.998869776725769, "rewards/meter/std": 0.00026758189778774977, "rewards/count_adherence/mean": 0.7734375, "rewards/count_adherence/std": 0.04650149121880531, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.946873664855957, "rewards/repeat_penalty/std": 0.06902104616165161, "rewards/total_composite/mean": 0.7318886518478394, "rewards/total_composite/std": 0.0725080594420433, "reward": 0.7318886518478394, "reward_std": 0.07250804454088211, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03919937461614609, "sampling/sampling_logp_difference/max": 1.6993675231933594, "sampling/importance_sampling_ratio/min": 0.1827991008758545, "sampling/importance_sampling_ratio/mean": 1.009418249130249, "sampling/importance_sampling_ratio/max": 1.9875471591949463, "entropy": 0.35818639770150185, "clip_ratio/low_mean": 0.0056649468606337905, "clip_ratio/low_min": 0.0056649468606337905, "clip_ratio/high_mean": 0.020963466726243496, "clip_ratio/high_max": 0.020963466726243496, "clip_ratio/region_mean": 0.026628413586877286, "reward_total_mean": 0.7318886518478394, "reward_meter_mean": 0.998869776725769, "reward_meter_std": 0.00026758189778774977, "reward_count_adherence_mean": 0.7734375, "reward_count_adherence_std": 0.04650149121880531, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.946873664855957, "reward_repeat_penalty_std": 0.06902104616165161, "reward_total_composite_mean": 0.7318886518478394, "reward_total_composite_std": 0.0725080594420433} {"timestamp_utc": "2026-04-12T02:32:26Z", "mode": "train", "global_step": 2688, "epoch": 0.10796481503795638, "loss": 0.0052, "grad_norm": 3.9790732860565186, "learning_rate": 1.8575757575757578e-06, "num_tokens": 6092497.0, "completions/mean_length": 140.125, "completions/min_length": 137.0, "completions/max_length": 144.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 140.125, "completions/min_terminated_length": 137.0, "completions/max_terminated_length": 144.0, "rewards/meter/mean": 0.9988832473754883, "rewards/meter/std": 0.0008235845598392189, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9988832473754883, "rewards/total_composite/std": 0.0008235845598392189, "reward": 0.9988832473754883, "reward_std": 0.0008235886925831437, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0495881512761116, "sampling/sampling_logp_difference/max": 2.3660759925842285, "sampling/importance_sampling_ratio/min": 0.09384826570749283, "sampling/importance_sampling_ratio/mean": 1.00178861618042, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39379359781742096, "clip_ratio/low_mean": 0.008036528481170535, "clip_ratio/low_min": 0.008036528481170535, "clip_ratio/high_mean": 0.04208028828725219, "clip_ratio/high_max": 0.04208028828725219, "clip_ratio/region_mean": 0.05011681676842272, "reward_total_mean": 0.9988832473754883, "reward_meter_mean": 0.9988832473754883, "reward_meter_std": 0.0008235845598392189, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9988832473754883, "reward_total_composite_std": 0.0008235845598392189} {"timestamp_utc": "2026-04-12T02:32:31Z", "mode": "train", "global_step": 2689, "epoch": 0.10800498051974133, "loss": 0.0078, "grad_norm": 3.656802177429199, "learning_rate": 1.8545454545454546e-06, "num_tokens": 6095097.0, "completions/mean_length": 132.0, "completions/min_length": 130.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.0, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.9993307590484619, "rewards/meter/std": 7.24185592844151e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.9636418223381042, "rewards/total_composite/std": 0.06610792875289917, "reward": 0.9636418223381042, "reward_std": 0.06610793620347977, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011180154979228973, "sampling/sampling_logp_difference/max": 0.804922342300415, "sampling/importance_sampling_ratio/min": 0.4471226632595062, "sampling/importance_sampling_ratio/mean": 0.9992504715919495, "sampling/importance_sampling_ratio/max": 1.8010199069976807, "entropy": 0.06141908373683691, "clip_ratio/low_mean": 0.004699247889220715, "clip_ratio/low_min": 0.004699247889220715, "clip_ratio/high_mean": 0.00855186500120908, "clip_ratio/high_max": 0.00855186500120908, "clip_ratio/region_mean": 0.013251112890429795, "reward_total_mean": 0.9636418223381042, "reward_meter_mean": 0.9993307590484619, "reward_meter_std": 7.24185592844151e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.9636418223381042, "reward_total_composite_std": 0.06610792875289917} {"timestamp_utc": "2026-04-12T02:32:35Z", "mode": "train", "global_step": 2690, "epoch": 0.10804514600152629, "loss": 0.0005, "grad_norm": 0.08232353627681732, "learning_rate": 1.8515151515151517e-06, "num_tokens": 6097105.0, "completions/mean_length": 67.0, "completions/min_length": 67.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981546401977539, "rewards/meter/std": 6.638248123636004e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981546401977539, "rewards/total_composite/std": 6.638248123636004e-06, "reward": 0.9981546401977539, "reward_std": 6.6471684476709925e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0016760394209995866, "sampling/sampling_logp_difference/max": 0.5546226501464844, "sampling/importance_sampling_ratio/min": 0.5742889642715454, "sampling/importance_sampling_ratio/mean": 0.9995981454849243, "sampling/importance_sampling_ratio/max": 1.0263193845748901, "entropy": 0.007268769608344883, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9981546401977539, "reward_meter_mean": 0.9981546401977539, "reward_meter_std": 6.638248123636004e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981546401977539, "reward_total_composite_std": 6.638248123636004e-06} {"timestamp_utc": "2026-04-12T02:32:41Z", "mode": "train", "global_step": 2691, "epoch": 0.10808531148331124, "loss": 0.0031, "grad_norm": 2.4885315895080566, "learning_rate": 1.8484848484848487e-06, "num_tokens": 6099876.0, "completions/mean_length": 166.375, "completions/min_length": 165.0, "completions/max_length": 168.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 166.375, "completions/min_terminated_length": 165.0, "completions/max_terminated_length": 168.0, "rewards/meter/mean": 0.9992619752883911, "rewards/meter/std": 0.00024342280812561512, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.9715026617050171, "rewards/total_composite/std": 0.051356665790081024, "reward": 0.9715026617050171, "reward_std": 0.05135667324066162, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015284853056073189, "sampling/sampling_logp_difference/max": 1.3873584270477295, "sampling/importance_sampling_ratio/min": 0.24973411858081818, "sampling/importance_sampling_ratio/mean": 1.0019655227661133, "sampling/importance_sampling_ratio/max": 1.893328070640564, "entropy": 0.07314449176192284, "clip_ratio/low_mean": 0.0015015150420367718, "clip_ratio/low_min": 0.0015015150420367718, "clip_ratio/high_mean": 0.009009360568597913, "clip_ratio/high_max": 0.009009360568597913, "clip_ratio/region_mean": 0.010510875610634685, "reward_total_mean": 0.9715026617050171, "reward_meter_mean": 0.9992619752883911, "reward_meter_std": 0.00024342280812561512, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_total_composite_mean": 0.9715026617050171, "reward_total_composite_std": 0.051356665790081024} {"timestamp_utc": "2026-04-12T02:32:46Z", "mode": "train", "global_step": 2692, "epoch": 0.1081254769650962, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.8454545454545455e-06, "num_tokens": 6101556.0, "completions/mean_length": 80.0, "completions/min_length": 80.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.0, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "reward": 0.7190229296684265, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002783673407975584, "sampling/sampling_logp_difference/max": 0.003993465099483728, "sampling/importance_sampling_ratio/min": 1.0, "sampling/importance_sampling_ratio/mean": 1.0002787113189697, "sampling/importance_sampling_ratio/max": 1.004001498222351, "entropy": 0.0021011842327425256, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7190229296684265, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:32:50Z", "mode": "train", "global_step": 2693, "epoch": 0.10816564244688115, "loss": 0.0009, "grad_norm": 1.4883558750152588, "learning_rate": 1.8424242424242426e-06, "num_tokens": 6103724.0, "completions/mean_length": 98.0, "completions/min_length": 98.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.0, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9992461204528809, "rewards/meter/std": 0.00010268352343700826, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992461204528809, "rewards/total_composite/std": 0.00010268352343700826, "reward": 0.9992461204528809, "reward_std": 0.00010269100312143564, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005682831630110741, "sampling/sampling_logp_difference/max": 1.3028526306152344, "sampling/importance_sampling_ratio/min": 0.6744450330734253, "sampling/importance_sampling_ratio/mean": 1.0015344619750977, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.01480487291701138, "clip_ratio/low_mean": 0.0038265305338427424, "clip_ratio/low_min": 0.0038265305338427424, "clip_ratio/high_mean": 0.0025510203558951616, "clip_ratio/high_max": 0.0025510203558951616, "clip_ratio/region_mean": 0.006377550889737904, "reward_total_mean": 0.9992461204528809, "reward_meter_mean": 0.9992461204528809, "reward_meter_std": 0.00010268352343700826, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992461204528809, "reward_total_composite_std": 0.00010268352343700826} {"timestamp_utc": "2026-04-12T02:32:55Z", "mode": "train", "global_step": 2694, "epoch": 0.1082058079286661, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.8393939393939394e-06, "num_tokens": 6105620.0, "completions/mean_length": 80.0, "completions/min_length": 80.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.0, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "reward": 0.7190229296684265, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002477174566593021, "sampling/sampling_logp_difference/max": 0.006054490804672241, "sampling/importance_sampling_ratio/min": 0.9999150037765503, "sampling/importance_sampling_ratio/mean": 1.000247597694397, "sampling/importance_sampling_ratio/max": 1.0060728788375854, "entropy": 0.0020191115036141127, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7190229296684265, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:32:59Z", "mode": "train", "global_step": 2695, "epoch": 0.10824597341045106, "loss": 0.064, "grad_norm": 12.76325798034668, "learning_rate": 1.8363636363636365e-06, "num_tokens": 6107356.0, "completions/mean_length": 63.0, "completions/min_length": 58.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.0, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.6803799867630005, "rewards/meter/std": 0.3893775939941406, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6803799867630005, "rewards/total_composite/std": 0.3893775939941406, "reward": 0.6803799867630005, "reward_std": 0.3893775939941406, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09513663500547409, "sampling/sampling_logp_difference/max": 2.7315380573272705, "sampling/importance_sampling_ratio/min": 0.06511905789375305, "sampling/importance_sampling_ratio/mean": 1.008878231048584, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.356488361954689, "clip_ratio/low_mean": 0.03159340703859925, "clip_ratio/low_min": 0.03159340703859925, "clip_ratio/high_mean": 0.0434071971103549, "clip_ratio/high_max": 0.0434071971103549, "clip_ratio/region_mean": 0.07500060414895415, "reward_total_mean": 0.6803799867630005, "reward_meter_mean": 0.6803799867630005, "reward_meter_std": 0.3893775939941406, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6803799867630005, "reward_total_composite_std": 0.3893775939941406} {"timestamp_utc": "2026-04-12T02:33:04Z", "mode": "train", "global_step": 2696, "epoch": 0.10828613889223601, "loss": 0.0176, "grad_norm": 9.303409576416016, "learning_rate": 1.8333333333333333e-06, "num_tokens": 6109361.0, "completions/mean_length": 66.625, "completions/min_length": 65.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.625, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9331727027893066, "rewards/meter/std": 0.008176974020898342, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9331727027893066, "rewards/total_composite/std": 0.008176974020898342, "reward": 0.9331727027893066, "reward_std": 0.008176976814866066, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.036672260612249374, "sampling/sampling_logp_difference/max": 0.9801895618438721, "sampling/importance_sampling_ratio/min": 0.3752399682998657, "sampling/importance_sampling_ratio/mean": 0.9995071291923523, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.18041662964969873, "clip_ratio/low_mean": 0.01482684025540948, "clip_ratio/low_min": 0.01482684025540948, "clip_ratio/high_mean": 0.016798642929643393, "clip_ratio/high_max": 0.016798642929643393, "clip_ratio/region_mean": 0.03162548318505287, "reward_total_mean": 0.9331727027893066, "reward_meter_mean": 0.9331727027893066, "reward_meter_std": 0.008176974020898342, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9331727027893066, "reward_total_composite_std": 0.008176974020898342} {"timestamp_utc": "2026-04-12T02:33:10Z", "mode": "train", "global_step": 2697, "epoch": 0.10832630437402097, "loss": 0.0029, "grad_norm": 0.7284796237945557, "learning_rate": 1.8303030303030305e-06, "num_tokens": 6112233.0, "completions/mean_length": 168.0, "completions/min_length": 166.0, "completions/max_length": 170.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 168.0, "completions/min_terminated_length": 166.0, "completions/max_terminated_length": 170.0, "rewards/meter/mean": 0.9992363452911377, "rewards/meter/std": 9.161681373370811e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992363452911377, "rewards/total_composite/std": 9.161681373370811e-05, "reward": 0.9992363452911377, "reward_std": 9.160472109215334e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007861536927521229, "sampling/sampling_logp_difference/max": 0.685643196105957, "sampling/importance_sampling_ratio/min": 0.5037661194801331, "sampling/importance_sampling_ratio/mean": 1.0003851652145386, "sampling/importance_sampling_ratio/max": 1.376589298248291, "entropy": 0.0484649995341897, "clip_ratio/low_mean": 0.002214635896962136, "clip_ratio/low_min": 0.002214635896962136, "clip_ratio/high_mean": 0.003729202609974891, "clip_ratio/high_max": 0.003729202609974891, "clip_ratio/region_mean": 0.005943838506937027, "reward_total_mean": 0.9992363452911377, "reward_meter_mean": 0.9992363452911377, "reward_meter_std": 9.161681373370811e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992363452911377, "reward_total_composite_std": 9.161681373370811e-05} {"timestamp_utc": "2026-04-12T02:33:14Z", "mode": "train", "global_step": 2698, "epoch": 0.10836646985580592, "loss": -0.0019, "grad_norm": 1.4812349081039429, "learning_rate": 1.8272727272727276e-06, "num_tokens": 6114086.0, "completions/mean_length": 71.625, "completions/min_length": 70.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.625, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.999380350112915, "rewards/meter/std": 0.00012280534429010004, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999380350112915, "rewards/total_composite/std": 0.00012280534429010004, "reward": 0.999380350112915, "reward_std": 0.0001228114851983264, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012132828123867512, "sampling/sampling_logp_difference/max": 1.2307639122009277, "sampling/importance_sampling_ratio/min": 0.29206937551498413, "sampling/importance_sampling_ratio/mean": 0.9971534013748169, "sampling/importance_sampling_ratio/max": 1.2945717573165894, "entropy": 0.051547068171203136, "clip_ratio/low_mean": 0.0035714285913854837, "clip_ratio/low_min": 0.0035714285913854837, "clip_ratio/high_mean": 0.00870500784367323, "clip_ratio/high_max": 0.00870500784367323, "clip_ratio/region_mean": 0.012276436435058713, "reward_total_mean": 0.999380350112915, "reward_meter_mean": 0.999380350112915, "reward_meter_std": 0.00012280534429010004, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.999380350112915, "reward_total_composite_std": 0.00012280534429010004} {"timestamp_utc": "2026-04-12T02:33:19Z", "mode": "train", "global_step": 2699, "epoch": 0.10840663533759087, "loss": 0.0187, "grad_norm": 21.836759567260742, "learning_rate": 1.8242424242424244e-06, "num_tokens": 6115559.0, "completions/mean_length": 34.125, "completions/min_length": 34.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.125, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9915792942047119, "rewards/meter/std": 0.0012008182238787413, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9915792942047119, "rewards/total_composite/std": 0.0012008182238787413, "reward": 0.9915792942047119, "reward_std": 0.001200812985189259, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024363644421100616, "sampling/sampling_logp_difference/max": 1.109255313873291, "sampling/importance_sampling_ratio/min": 0.3298044800758362, "sampling/importance_sampling_ratio/mean": 1.0031808614730835, "sampling/importance_sampling_ratio/max": 1.8209221363067627, "entropy": 0.18179897591471672, "clip_ratio/low_mean": 0.01460084063000977, "clip_ratio/low_min": 0.01460084063000977, "clip_ratio/high_mean": 0.007352941203862429, "clip_ratio/high_max": 0.007352941203862429, "clip_ratio/region_mean": 0.0219537818338722, "reward_total_mean": 0.9915792942047119, "reward_meter_mean": 0.9915792942047119, "reward_meter_std": 0.0012008182238787413, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9915792942047119, "reward_total_composite_std": 0.0012008182238787413} {"timestamp_utc": "2026-04-12T02:33:23Z", "mode": "train", "global_step": 2700, "epoch": 0.10844680081937583, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.8212121212121215e-06, "num_tokens": 6117340.0, "completions/mean_length": 71.625, "completions/min_length": 71.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.625, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.99944007396698, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.99944007396698, "rewards/total_composite/std": 0.0, "reward": 0.99944007396698, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.004740848205983639, "sampling/sampling_logp_difference/max": 0.2854344844818115, "sampling/importance_sampling_ratio/min": 0.7516875863075256, "sampling/importance_sampling_ratio/mean": 1.002069115638733, "sampling/importance_sampling_ratio/max": 1.2885301113128662, "entropy": 0.03858046652749181, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.99944007396698, "reward_meter_mean": 0.99944007396698, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.99944007396698, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:34:38Z", "mode": "eval", "global_step": 2700, "epoch": 0.10844680081937583, "eval_loss": NaN, "eval_runtime": 74.8526, "eval_samples_per_second": 1.389, "eval_steps_per_second": 0.174, "eval_num_tokens": 6117340.0, "eval_completions/mean_length": 205.91346153846155, "eval_completions/min_length": 61.07692307692308, "eval_completions/max_length": 402.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 205.91346153846155, "eval_completions/min_terminated_length": 61.07692307692308, "eval_completions/max_terminated_length": 402.0, "eval_rewards/meter/mean": 0.818766135435838, "eval_rewards/meter/std": 0.28128915272939664, "eval_rewards/count_adherence/mean": 0.9412473761118375, "eval_rewards/count_adherence/std": 0.08066883425299938, "eval_rewards/arabic_clean/mean": 0.9807692307692307, "eval_rewards/arabic_clean/std": 0.05439282839114849, "eval_rewards/repeat_penalty/mean": 0.917067867058974, "eval_rewards/repeat_penalty/std": 0.09873221929256733, "eval_rewards/total_composite/mean": 0.7056683852122381, "eval_rewards/total_composite/std": 0.3022650325527558, "eval_reward": 0.7056683852122381, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.026865268770891886, "eval_sampling/sampling_logp_difference/max": 1.0165979678814228, "eval_sampling/importance_sampling_ratio/min": 0.36844213765401107, "eval_sampling/importance_sampling_ratio/mean": 1.00816364471729, "eval_sampling/importance_sampling_ratio/max": 1.4687065803087676, "eval_entropy": 0.2846654344063539, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7056683852122381, "eval_reward_meter_mean": 0.818766135435838, "eval_reward_meter_std": 0.28128915272939664, "eval_reward_count_adherence_mean": 0.9412473761118375, "eval_reward_count_adherence_std": 0.08066883425299938, "eval_reward_arabic_clean_mean": 0.9807692307692307, "eval_reward_arabic_clean_std": 0.05439282839114849, "eval_reward_repeat_penalty_mean": 0.917067867058974, "eval_reward_repeat_penalty_std": 0.09873221929256733, "eval_reward_total_composite_mean": 0.7056683852122381, "eval_reward_total_composite_std": 0.3022650325527558} {"timestamp_utc": "2026-04-12T02:34:45Z", "mode": "train", "global_step": 2701, "epoch": 0.10848696630116078, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.8181818181818183e-06, "num_tokens": 6119068.0, "completions/mean_length": 67.0, "completions/min_length": 67.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981522560119629, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981522560119629, "rewards/total_composite/std": 0.0, "reward": 0.9981522560119629, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0014518008101731539, "sampling/sampling_logp_difference/max": 0.2512565851211548, "sampling/importance_sampling_ratio/min": 0.7778228521347046, "sampling/importance_sampling_ratio/mean": 0.9999621510505676, "sampling/importance_sampling_ratio/max": 1.0299800634384155, "entropy": 0.012467616703361273, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9981522560119629, "reward_meter_mean": 0.9981522560119629, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981522560119629, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:34:50Z", "mode": "train", "global_step": 2702, "epoch": 0.10852713178294573, "loss": -0.001, "grad_norm": 3.3459887504577637, "learning_rate": 1.8151515151515153e-06, "num_tokens": 6120768.0, "completions/mean_length": 67.5, "completions/min_length": 65.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.5, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.937604546546936, "rewards/meter/std": 0.136740043759346, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.937604546546936, "rewards/total_composite/std": 0.136740043759346, "reward": 0.937604546546936, "reward_std": 0.1367400586605072, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025002291426062584, "sampling/sampling_logp_difference/max": 1.505655288696289, "sampling/importance_sampling_ratio/min": 0.22187186777591705, "sampling/importance_sampling_ratio/mean": 1.0090194940567017, "sampling/importance_sampling_ratio/max": 1.6750621795654297, "entropy": 0.19097852148115635, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/high_mean": 0.007548359571956098, "clip_ratio/high_max": 0.007548359571956098, "clip_ratio/region_mean": 0.00941403117030859, "reward_total_mean": 0.937604546546936, "reward_meter_mean": 0.937604546546936, "reward_meter_std": 0.136740043759346, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.937604546546936, "reward_total_composite_std": 0.136740043759346} {"timestamp_utc": "2026-04-12T02:34:56Z", "mode": "train", "global_step": 2703, "epoch": 0.10856729726473069, "loss": 0.0104, "grad_norm": 2.7519359588623047, "learning_rate": 1.8121212121212124e-06, "num_tokens": 6124022.0, "completions/mean_length": 213.75, "completions/min_length": 207.0, "completions/max_length": 218.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 213.75, "completions/min_terminated_length": 207.0, "completions/max_terminated_length": 218.0, "rewards/meter/mean": 0.9976348876953125, "rewards/meter/std": 0.0003981303598266095, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9711538553237915, "rewards/repeat_penalty/std": 0.039811473339796066, "rewards/total_composite/mean": 0.9688572883605957, "rewards/total_composite/std": 0.039728421717882156, "reward": 0.9688572883605957, "reward_std": 0.03972842916846275, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.034267134964466095, "sampling/sampling_logp_difference/max": 1.2862721681594849, "sampling/importance_sampling_ratio/min": 0.276298850774765, "sampling/importance_sampling_ratio/mean": 1.0066944360733032, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.220721784979105, "clip_ratio/low_mean": 0.016337704844772816, "clip_ratio/low_min": 0.016337704844772816, "clip_ratio/high_mean": 0.019398832926526666, "clip_ratio/high_max": 0.019398832926526666, "clip_ratio/region_mean": 0.03573653777129948, "reward_total_mean": 0.9688572883605957, "reward_meter_mean": 0.9976348876953125, "reward_meter_std": 0.0003981303598266095, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9711538553237915, "reward_repeat_penalty_std": 0.039811473339796066, "reward_total_composite_mean": 0.9688572883605957, "reward_total_composite_std": 0.039728421717882156} {"timestamp_utc": "2026-04-12T02:35:01Z", "mode": "train", "global_step": 2704, "epoch": 0.10860746274651564, "loss": -0.0003, "grad_norm": 0.004991916939616203, "learning_rate": 1.8090909090909092e-06, "num_tokens": 6125799.0, "completions/mean_length": 67.125, "completions/min_length": 67.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.125, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9981493353843689, "rewards/meter/std": 8.239712769864127e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981493353843689, "rewards/total_composite/std": 8.239712769864127e-06, "reward": 0.9981493353843689, "reward_std": 8.248746780736838e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0028709208127111197, "sampling/sampling_logp_difference/max": 0.8200433254241943, "sampling/importance_sampling_ratio/min": 0.44041258096694946, "sampling/importance_sampling_ratio/mean": 0.9988438487052917, "sampling/importance_sampling_ratio/max": 1.0250744819641113, "entropy": 0.007660032075364143, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/high_mean": 0.0018382353009656072, "clip_ratio/high_max": 0.0018382353009656072, "clip_ratio/region_mean": 0.003703906899318099, "reward_total_mean": 0.9981493353843689, "reward_meter_mean": 0.9981493353843689, "reward_meter_std": 8.239712769864127e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981493353843689, "reward_total_composite_std": 8.239712769864127e-06} {"timestamp_utc": "2026-04-12T02:35:05Z", "mode": "train", "global_step": 2705, "epoch": 0.1086476282283006, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.8060606060606063e-06, "num_tokens": 6127407.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00024437991669401526, "sampling/sampling_logp_difference/max": 0.002418494550511241, "sampling/importance_sampling_ratio/min": 0.9996000528335571, "sampling/importance_sampling_ratio/mean": 1.0002423524856567, "sampling/importance_sampling_ratio/max": 1.002421498298645, "entropy": 0.0020572251814883202, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:35:09Z", "mode": "train", "global_step": 2706, "epoch": 0.10868779371008555, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.803030303030303e-06, "num_tokens": 6128951.0, "completions/mean_length": 32.0, "completions/min_length": 32.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "reward": 0.9992982149124146, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002643416519276798, "sampling/sampling_logp_difference/max": 0.011745302006602287, "sampling/importance_sampling_ratio/min": 0.9924525022506714, "sampling/importance_sampling_ratio/mean": 1.0001347064971924, "sampling/importance_sampling_ratio/max": 1.0118145942687988, "entropy": 0.002913154661655426, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9992982149124146, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:35:14Z", "mode": "train", "global_step": 2707, "epoch": 0.1087279591918705, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.8000000000000001e-06, "num_tokens": 6130687.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00021008482144679874, "sampling/sampling_logp_difference/max": 0.0022089765407145023, "sampling/importance_sampling_ratio/min": 0.9998108148574829, "sampling/importance_sampling_ratio/mean": 1.0002092123031616, "sampling/importance_sampling_ratio/max": 1.0022114515304565, "entropy": 0.001665496762143448, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:35:19Z", "mode": "train", "global_step": 2708, "epoch": 0.10876812467365546, "loss": -0.0031, "grad_norm": 1.1375532150268555, "learning_rate": 1.796969696969697e-06, "num_tokens": 6132899.0, "completions/mean_length": 106.5, "completions/min_length": 105.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.5, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.999268651008606, "rewards/meter/std": 0.00017192552331835032, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999268651008606, "rewards/total_composite/std": 0.00017192552331835032, "reward": 0.999268651008606, "reward_std": 0.00017192917584907264, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008916010148823261, "sampling/sampling_logp_difference/max": 0.8204793930053711, "sampling/importance_sampling_ratio/min": 0.44022059440612793, "sampling/importance_sampling_ratio/mean": 1.0018383264541626, "sampling/importance_sampling_ratio/max": 1.3604763746261597, "entropy": 0.053598769940435886, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.004673305433243513, "clip_ratio/high_max": 0.004673305433243513, "clip_ratio/region_mean": 0.004673305433243513, "reward_total_mean": 0.999268651008606, "reward_meter_mean": 0.999268651008606, "reward_meter_std": 0.00017192552331835032, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.999268651008606, "reward_total_composite_std": 0.00017192552331835032} {"timestamp_utc": "2026-04-12T02:35:24Z", "mode": "train", "global_step": 2709, "epoch": 0.10880829015544041, "loss": 0.0051, "grad_norm": 3.0748214721679688, "learning_rate": 1.793939393939394e-06, "num_tokens": 6135215.0, "completions/mean_length": 102.5, "completions/min_length": 98.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.5, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.998659610748291, "rewards/meter/std": 0.0006201548385433853, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998659610748291, "rewards/total_composite/std": 0.0006201548385433853, "reward": 0.998659610748291, "reward_std": 0.0006201544310897589, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03828549385070801, "sampling/sampling_logp_difference/max": 1.1598763465881348, "sampling/importance_sampling_ratio/min": 0.35953280329704285, "sampling/importance_sampling_ratio/mean": 1.0111318826675415, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2976728491485119, "clip_ratio/low_mean": 0.010879121022298932, "clip_ratio/low_min": 0.010879121022298932, "clip_ratio/high_mean": 0.01722477504517883, "clip_ratio/high_max": 0.01722477504517883, "clip_ratio/region_mean": 0.028103896067477763, "reward_total_mean": 0.998659610748291, "reward_meter_mean": 0.998659610748291, "reward_meter_std": 0.0006201548385433853, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.998659610748291, "reward_total_composite_std": 0.0006201548385433853} {"timestamp_utc": "2026-04-12T02:35:29Z", "mode": "train", "global_step": 2710, "epoch": 0.10884845563722537, "loss": 0.0027, "grad_norm": 1.891419529914856, "learning_rate": 1.7909090909090908e-06, "num_tokens": 6137547.0, "completions/mean_length": 142.5, "completions/min_length": 141.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 142.5, "completions/min_terminated_length": 141.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.999262809753418, "rewards/meter/std": 0.00011097361129941419, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8392857313156128, "rewards/repeat_penalty/std": 0.09155284613370895, "rewards/total_composite/mean": 0.8386699557304382, "rewards/total_composite/std": 0.0915173664689064, "reward": 0.8386699557304382, "reward_std": 0.0915173590183258, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010055704973638058, "sampling/sampling_logp_difference/max": 0.6665773391723633, "sampling/importance_sampling_ratio/min": 0.5134629607200623, "sampling/importance_sampling_ratio/mean": 1.0032413005828857, "sampling/importance_sampling_ratio/max": 1.472945213317871, "entropy": 0.0695994533598423, "clip_ratio/low_mean": 0.003502913983538747, "clip_ratio/low_min": 0.003502913983538747, "clip_ratio/high_mean": 0.001754407538101077, "clip_ratio/high_max": 0.001754407538101077, "clip_ratio/region_mean": 0.005257321521639824, "reward_total_mean": 0.8386699557304382, "reward_meter_mean": 0.999262809753418, "reward_meter_std": 0.00011097361129941419, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8392857313156128, "reward_repeat_penalty_std": 0.09155284613370895, "reward_total_composite_mean": 0.8386699557304382, "reward_total_composite_std": 0.0915173664689064} {"timestamp_utc": "2026-04-12T02:35:34Z", "mode": "train", "global_step": 2711, "epoch": 0.10888862111901032, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.787878787878788e-06, "num_tokens": 6139309.0, "completions/mean_length": 66.25, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.25, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981522560119629, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981522560119629, "rewards/total_composite/std": 0.0, "reward": 0.9981522560119629, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0032224601600319147, "sampling/sampling_logp_difference/max": 0.3331582546234131, "sampling/importance_sampling_ratio/min": 0.729793906211853, "sampling/importance_sampling_ratio/mean": 1.0023350715637207, "sampling/importance_sampling_ratio/max": 1.395367980003357, "entropy": 0.016254089423455298, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9981522560119629, "reward_meter_mean": 0.9981522560119629, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981522560119629, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:35:38Z", "mode": "train", "global_step": 2712, "epoch": 0.10892878660079527, "loss": -0.0068, "grad_norm": 3.4030656814575195, "learning_rate": 1.7848484848484851e-06, "num_tokens": 6141147.0, "completions/mean_length": 60.75, "completions/min_length": 59.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.75, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9972513318061829, "rewards/meter/std": 0.0001800779573386535, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9972513318061829, "rewards/total_composite/std": 0.0001800779573386535, "reward": 0.9972513318061829, "reward_std": 0.00018008207553066313, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0070107607170939445, "sampling/sampling_logp_difference/max": 0.9685866832733154, "sampling/importance_sampling_ratio/min": 0.7316676378250122, "sampling/importance_sampling_ratio/mean": 1.0053867101669312, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.02895409008488059, "clip_ratio/low_mean": 0.0020491802133619785, "clip_ratio/low_min": 0.0020491802133619785, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0020491802133619785, "reward_total_mean": 0.9972513318061829, "reward_meter_mean": 0.9972513318061829, "reward_meter_std": 0.0001800779573386535, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9972513318061829, "reward_total_composite_std": 0.0001800779573386535} {"timestamp_utc": "2026-04-12T02:35:43Z", "mode": "train", "global_step": 2713, "epoch": 0.10896895208258023, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.781818181818182e-06, "num_tokens": 6142939.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0010938982013612986, "sampling/sampling_logp_difference/max": 0.2828308939933777, "sampling/importance_sampling_ratio/min": 0.7536472678184509, "sampling/importance_sampling_ratio/mean": 0.9994007349014282, "sampling/importance_sampling_ratio/max": 1.023779034614563, "entropy": 0.004417360032675788, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:35:47Z", "mode": "train", "global_step": 2714, "epoch": 0.10900911756436518, "loss": 0.0083, "grad_norm": 2.3612427711486816, "learning_rate": 1.778787878787879e-06, "num_tokens": 6144770.0, "completions/mean_length": 78.875, "completions/min_length": 77.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.875, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9988387227058411, "rewards/meter/std": 0.0002311533025931567, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9988387227058411, "rewards/total_composite/std": 0.0002311533025931567, "reward": 0.9988387227058411, "reward_std": 0.00023116436204873025, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028173334896564484, "sampling/sampling_logp_difference/max": 1.179065227508545, "sampling/importance_sampling_ratio/min": 0.3075661063194275, "sampling/importance_sampling_ratio/mean": 1.0010093450546265, "sampling/importance_sampling_ratio/max": 1.6622776985168457, "entropy": 0.15530052781105042, "clip_ratio/low_mean": 0.0015432098880410194, "clip_ratio/low_min": 0.0015432098880410194, "clip_ratio/high_mean": 0.016068846685811877, "clip_ratio/high_max": 0.016068846685811877, "clip_ratio/region_mean": 0.017612056573852897, "reward_total_mean": 0.9988387227058411, "reward_meter_mean": 0.9988387227058411, "reward_meter_std": 0.0002311533025931567, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9988387227058411, "reward_total_composite_std": 0.0002311533025931567} {"timestamp_utc": "2026-04-12T02:35:53Z", "mode": "train", "global_step": 2715, "epoch": 0.10904928304615014, "loss": 0.0025, "grad_norm": 0.7956473231315613, "learning_rate": 1.775757575757576e-06, "num_tokens": 6147492.0, "completions/mean_length": 159.25, "completions/min_length": 159.0, "completions/max_length": 160.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 159.25, "completions/min_terminated_length": 159.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.9979335069656372, "rewards/meter/std": 1.7607557310839184e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9027777910232544, "rewards/repeat_penalty/std": 0.03928370773792267, "rewards/total_composite/mean": 0.9009125232696533, "rewards/total_composite/std": 0.03921113535761833, "reward": 0.9009125232696533, "reward_std": 0.03921113163232803, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009351929649710655, "sampling/sampling_logp_difference/max": 0.5963006019592285, "sampling/importance_sampling_ratio/min": 0.5508456826210022, "sampling/importance_sampling_ratio/mean": 1.0039114952087402, "sampling/importance_sampling_ratio/max": 1.4207649230957031, "entropy": 0.08590239379554987, "clip_ratio/low_mean": 0.0015674135065637529, "clip_ratio/low_min": 0.0015674135065637529, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0015674135065637529, "reward_total_mean": 0.9009125232696533, "reward_meter_mean": 0.9979335069656372, "reward_meter_std": 1.7607557310839184e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9027777910232544, "reward_repeat_penalty_std": 0.03928370773792267, "reward_total_composite_mean": 0.9009125232696533, "reward_total_composite_std": 0.03921113535761833} {"timestamp_utc": "2026-04-12T02:36:02Z", "mode": "train", "global_step": 2716, "epoch": 0.10908944852793509, "loss": -0.0565, "grad_norm": 1.8313432931900024, "learning_rate": 1.7727272727272729e-06, "num_tokens": 6152509.0, "completions/mean_length": 438.125, "completions/min_length": 361.0, "completions/max_length": 459.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 438.125, "completions/min_terminated_length": 361.0, "completions/max_terminated_length": 459.0, "rewards/meter/mean": 0.9989956617355347, "rewards/meter/std": 0.00016425758076366037, "rewards/count_adherence/mean": 0.7666666507720947, "rewards/count_adherence/std": 0.07126966118812561, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9524672031402588, "rewards/repeat_penalty/std": 0.05135857313871384, "rewards/total_composite/mean": 0.7314679622650146, "rewards/total_composite/std": 0.09445607662200928, "reward": 0.7314679622650146, "reward_std": 0.09445606172084808, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04289248585700989, "sampling/sampling_logp_difference/max": 1.8779144287109375, "sampling/importance_sampling_ratio/min": 0.15290868282318115, "sampling/importance_sampling_ratio/mean": 1.0078824758529663, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3533891662955284, "clip_ratio/low_mean": 0.008343940833583474, "clip_ratio/low_min": 0.008343940833583474, "clip_ratio/high_mean": 0.014018616639077663, "clip_ratio/high_max": 0.014018616639077663, "clip_ratio/region_mean": 0.022362557472661138, "reward_total_mean": 0.7314679622650146, "reward_meter_mean": 0.9989956617355347, "reward_meter_std": 0.00016425758076366037, "reward_count_adherence_mean": 0.7666666507720947, "reward_count_adherence_std": 0.07126966118812561, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9524672031402588, "reward_repeat_penalty_std": 0.05135857313871384, "reward_total_composite_mean": 0.7314679622650146, "reward_total_composite_std": 0.09445607662200928} {"timestamp_utc": "2026-04-12T02:36:08Z", "mode": "train", "global_step": 2717, "epoch": 0.10912961400972004, "loss": 0.0033, "grad_norm": 1.0715020895004272, "learning_rate": 1.76969696969697e-06, "num_tokens": 6155040.0, "completions/mean_length": 131.375, "completions/min_length": 130.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.375, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.9993519186973572, "rewards/meter/std": 4.195739529677667e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993519186973572, "rewards/total_composite/std": 4.195739529677667e-05, "reward": 0.9993519186973572, "reward_std": 4.196614827378653e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011481878347694874, "sampling/sampling_logp_difference/max": 1.0313186645507812, "sampling/importance_sampling_ratio/min": 0.35653650760650635, "sampling/importance_sampling_ratio/mean": 0.9986187815666199, "sampling/importance_sampling_ratio/max": 1.3561691045761108, "entropy": 0.05451698787510395, "clip_ratio/low_mean": 0.0056604581186547875, "clip_ratio/low_min": 0.0056604581186547875, "clip_ratio/high_mean": 0.00957868475234136, "clip_ratio/high_max": 0.00957868475234136, "clip_ratio/region_mean": 0.015239142870996147, "reward_total_mean": 0.9993519186973572, "reward_meter_mean": 0.9993519186973572, "reward_meter_std": 4.195739529677667e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993519186973572, "reward_total_composite_std": 4.195739529677667e-05} {"timestamp_utc": "2026-04-12T02:36:15Z", "mode": "train", "global_step": 2718, "epoch": 0.109169779491505, "loss": 0.0052, "grad_norm": 3.4733986854553223, "learning_rate": 1.7666666666666668e-06, "num_tokens": 6159098.0, "completions/mean_length": 307.25, "completions/min_length": 298.0, "completions/max_length": 313.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 307.25, "completions/min_terminated_length": 298.0, "completions/max_terminated_length": 313.0, "rewards/meter/mean": 0.9986767768859863, "rewards/meter/std": 0.0004087568959221244, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9632352590560913, "rewards/repeat_penalty/std": 0.04376610368490219, "rewards/total_composite/mean": 0.8657643795013428, "rewards/total_composite/std": 0.039332009851932526, "reward": 0.8657643795013428, "reward_std": 0.03933200240135193, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.059995777904987335, "sampling/sampling_logp_difference/max": 3.847691059112549, "sampling/importance_sampling_ratio/min": 0.02132892794907093, "sampling/importance_sampling_ratio/mean": 1.010833740234375, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5477991737425327, "clip_ratio/low_mean": 0.023148135747760534, "clip_ratio/low_min": 0.023148135747760534, "clip_ratio/high_mean": 0.027308928780257702, "clip_ratio/high_max": 0.027308928780257702, "clip_ratio/region_mean": 0.050457064528018236, "reward_total_mean": 0.8657643795013428, "reward_meter_mean": 0.9986767768859863, "reward_meter_std": 0.0004087568959221244, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9632352590560913, "reward_repeat_penalty_std": 0.04376610368490219, "reward_total_composite_mean": 0.8657643795013428, "reward_total_composite_std": 0.039332009851932526} {"timestamp_utc": "2026-04-12T02:36:19Z", "mode": "train", "global_step": 2719, "epoch": 0.10920994497328995, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.7636363636363638e-06, "num_tokens": 6160514.0, "completions/mean_length": 29.0, "completions/min_length": 29.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9957436919212341, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957436919212341, "rewards/total_composite/std": 0.0, "reward": 0.9957436919212341, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 4.7315745177911595e-05, "sampling/sampling_logp_difference/max": 0.0008822012459859252, "sampling/importance_sampling_ratio/min": 0.9999879002571106, "sampling/importance_sampling_ratio/mean": 1.0000470876693726, "sampling/importance_sampling_ratio/max": 1.000882625579834, "entropy": 0.00040178035851567984, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9957436919212341, "reward_meter_mean": 0.9957436919212341, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9957436919212341, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:36:24Z", "mode": "train", "global_step": 2720, "epoch": 0.1092501104550749, "loss": 0.0004, "grad_norm": 0.7155827283859253, "learning_rate": 1.7606060606060606e-06, "num_tokens": 6162606.0, "completions/mean_length": 98.5, "completions/min_length": 98.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.5, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.9994127750396729, "rewards/meter/std": 2.115286042680964e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994127750396729, "rewards/total_composite/std": 2.115286042680964e-05, "reward": 0.9994127750396729, "reward_std": 2.114395101671107e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008277514949440956, "sampling/sampling_logp_difference/max": 0.9306962490081787, "sampling/importance_sampling_ratio/min": 0.3942791223526001, "sampling/importance_sampling_ratio/mean": 1.0007652044296265, "sampling/importance_sampling_ratio/max": 1.7030972242355347, "entropy": 0.03585473005659878, "clip_ratio/low_mean": 0.0024999999441206455, "clip_ratio/low_min": 0.0024999999441206455, "clip_ratio/high_mean": 0.005076272878795862, "clip_ratio/high_max": 0.005076272878795862, "clip_ratio/region_mean": 0.007576272822916508, "reward_total_mean": 0.9994127750396729, "reward_meter_mean": 0.9994127750396729, "reward_meter_std": 2.115286042680964e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994127750396729, "reward_total_composite_std": 2.115286042680964e-05} {"timestamp_utc": "2026-04-12T02:36:29Z", "mode": "train", "global_step": 2721, "epoch": 0.10929027593685986, "loss": 0.0456, "grad_norm": 5.66716194152832, "learning_rate": 1.7575757575757577e-06, "num_tokens": 6164482.0, "completions/mean_length": 70.5, "completions/min_length": 66.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.5, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9914366006851196, "rewards/meter/std": 0.004417366813868284, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9914366006851196, "rewards/total_composite/std": 0.004417366813868284, "reward": 0.9914366006851196, "reward_std": 0.004417361691594124, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.054880011826753616, "sampling/sampling_logp_difference/max": 1.349165916442871, "sampling/importance_sampling_ratio/min": 0.2594565749168396, "sampling/importance_sampling_ratio/mean": 0.998794436454773, "sampling/importance_sampling_ratio/max": 1.5672898292541504, "entropy": 0.3494097013026476, "clip_ratio/low_mean": 0.017159562092274427, "clip_ratio/low_min": 0.017159562092274427, "clip_ratio/high_mean": 0.02390979148913175, "clip_ratio/high_max": 0.02390979148913175, "clip_ratio/region_mean": 0.041069353581406176, "reward_total_mean": 0.9914366006851196, "reward_meter_mean": 0.9914366006851196, "reward_meter_std": 0.004417366813868284, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9914366006851196, "reward_total_composite_std": 0.004417366813868284} {"timestamp_utc": "2026-04-12T02:36:33Z", "mode": "train", "global_step": 2722, "epoch": 0.10933044141864481, "loss": 0.0176, "grad_norm": 11.017292976379395, "learning_rate": 1.7545454545454545e-06, "num_tokens": 6166347.0, "completions/mean_length": 69.125, "completions/min_length": 68.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.125, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9949984550476074, "rewards/meter/std": 0.011957395821809769, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9949984550476074, "rewards/total_composite/std": 0.011957395821809769, "reward": 0.9949984550476074, "reward_std": 0.01195740420371294, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05068688467144966, "sampling/sampling_logp_difference/max": 2.848238945007324, "sampling/importance_sampling_ratio/min": 0.05794627591967583, "sampling/importance_sampling_ratio/mean": 0.996598482131958, "sampling/importance_sampling_ratio/max": 1.5986580848693848, "entropy": 0.22193360701203346, "clip_ratio/low_mean": 0.005281690042465925, "clip_ratio/low_min": 0.005281690042465925, "clip_ratio/high_mean": 0.027233949513174593, "clip_ratio/high_max": 0.027233949513174593, "clip_ratio/region_mean": 0.03251563955564052, "reward_total_mean": 0.9949984550476074, "reward_meter_mean": 0.9949984550476074, "reward_meter_std": 0.011957395821809769, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9949984550476074, "reward_total_composite_std": 0.011957395821809769} {"timestamp_utc": "2026-04-12T02:36:40Z", "mode": "train", "global_step": 2723, "epoch": 0.10937060690042977, "loss": 0.0093, "grad_norm": 2.760967254638672, "learning_rate": 1.7515151515151516e-06, "num_tokens": 6169685.0, "completions/mean_length": 235.25, "completions/min_length": 225.0, "completions/max_length": 242.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 235.25, "completions/min_terminated_length": 225.0, "completions/max_terminated_length": 242.0, "rewards/meter/mean": 0.9979515671730042, "rewards/meter/std": 0.0015659787459298968, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9711538553237915, "rewards/repeat_penalty/std": 0.039811473339796066, "rewards/total_composite/mean": 0.969149649143219, "rewards/total_composite/std": 0.0393538773059845, "reward": 0.969149649143219, "reward_std": 0.0393538735806942, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05202290415763855, "sampling/sampling_logp_difference/max": 1.2681152820587158, "sampling/importance_sampling_ratio/min": 0.2813614308834076, "sampling/importance_sampling_ratio/mean": 1.0089517831802368, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5150004588067532, "clip_ratio/low_mean": 0.013798366067931056, "clip_ratio/low_min": 0.013798366067931056, "clip_ratio/high_mean": 0.03575678775086999, "clip_ratio/high_max": 0.03575678775086999, "clip_ratio/region_mean": 0.049555153818801045, "reward_total_mean": 0.969149649143219, "reward_meter_mean": 0.9979515671730042, "reward_meter_std": 0.0015659787459298968, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9711538553237915, "reward_repeat_penalty_std": 0.039811473339796066, "reward_total_composite_mean": 0.969149649143219, "reward_total_composite_std": 0.0393538773059845} {"timestamp_utc": "2026-04-12T02:36:45Z", "mode": "train", "global_step": 2724, "epoch": 0.10941077238221472, "loss": 0.0275, "grad_norm": 13.60006046295166, "learning_rate": 1.7484848484848486e-06, "num_tokens": 6171486.0, "completions/mean_length": 65.125, "completions/min_length": 61.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.125, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9050167798995972, "rewards/meter/std": 0.11941422522068024, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9050167798995972, "rewards/total_composite/std": 0.11941422522068024, "reward": 0.9050167798995972, "reward_std": 0.11941422522068024, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07930425554513931, "sampling/sampling_logp_difference/max": 1.9395782947540283, "sampling/importance_sampling_ratio/min": 0.14376457035541534, "sampling/importance_sampling_ratio/mean": 0.9989842772483826, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36046687327325344, "clip_ratio/low_mean": 0.014925372786819935, "clip_ratio/low_min": 0.014925372786819935, "clip_ratio/high_mean": 0.0387622892158106, "clip_ratio/high_max": 0.0387622892158106, "clip_ratio/region_mean": 0.05368766200263053, "reward_total_mean": 0.9050167798995972, "reward_meter_mean": 0.9050167798995972, "reward_meter_std": 0.11941422522068024, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9050167798995972, "reward_total_composite_std": 0.11941422522068024} {"timestamp_utc": "2026-04-12T02:36:49Z", "mode": "train", "global_step": 2725, "epoch": 0.10945093786399968, "loss": -0.0057, "grad_norm": 5.4172844886779785, "learning_rate": 1.7454545454545456e-06, "num_tokens": 6173408.0, "completions/mean_length": 70.25, "completions/min_length": 68.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.25, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9948872923851013, "rewards/meter/std": 0.0014726987574249506, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9948872923851013, "rewards/total_composite/std": 0.0014726987574249506, "reward": 0.9948872923851013, "reward_std": 0.0014726977096870542, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04584960639476776, "sampling/sampling_logp_difference/max": 1.2650136947631836, "sampling/importance_sampling_ratio/min": 0.28223544359207153, "sampling/importance_sampling_ratio/mean": 1.0125356912612915, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2636737637221813, "clip_ratio/low_mean": 0.01608805637806654, "clip_ratio/low_min": 0.01608805637806654, "clip_ratio/high_mean": 0.012341346475295722, "clip_ratio/high_max": 0.012341346475295722, "clip_ratio/region_mean": 0.028429402853362262, "reward_total_mean": 0.9948872923851013, "reward_meter_mean": 0.9948872923851013, "reward_meter_std": 0.0014726987574249506, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9948872923851013, "reward_total_composite_std": 0.0014726987574249506} {"timestamp_utc": "2026-04-12T02:36:54Z", "mode": "train", "global_step": 2726, "epoch": 0.10949110334578463, "loss": -0.004, "grad_norm": 2.82613468170166, "learning_rate": 1.7424242424242427e-06, "num_tokens": 6175647.0, "completions/mean_length": 103.875, "completions/min_length": 102.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.875, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.9729191064834595, "rewards/meter/std": 0.010998339392244816, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9729191064834595, "rewards/total_composite/std": 0.010998339392244816, "reward": 0.9729191064834595, "reward_std": 0.010998345911502838, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0373404435813427, "sampling/sampling_logp_difference/max": 1.3049564361572266, "sampling/importance_sampling_ratio/min": 0.27118435502052307, "sampling/importance_sampling_ratio/mean": 1.0128881931304932, "sampling/importance_sampling_ratio/max": 1.7675621509552002, "entropy": 0.3632195107638836, "clip_ratio/low_mean": 0.01458781841211021, "clip_ratio/low_min": 0.01458781841211021, "clip_ratio/high_mean": 0.009652552893385291, "clip_ratio/high_max": 0.009652552893385291, "clip_ratio/region_mean": 0.0242403713054955, "reward_total_mean": 0.9729191064834595, "reward_meter_mean": 0.9729191064834595, "reward_meter_std": 0.010998339392244816, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9729191064834595, "reward_total_composite_std": 0.010998339392244816} {"timestamp_utc": "2026-04-12T02:36:59Z", "mode": "train", "global_step": 2727, "epoch": 0.10953126882756958, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.7393939393939397e-06, "num_tokens": 6177392.0, "completions/mean_length": 71.125, "completions/min_length": 71.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.125, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.99944007396698, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.99944007396698, "rewards/total_composite/std": 0.0, "reward": 0.99944007396698, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.004035182762891054, "sampling/sampling_logp_difference/max": 0.31750011444091797, "sampling/importance_sampling_ratio/min": 0.7279666066169739, "sampling/importance_sampling_ratio/mean": 1.0021454095840454, "sampling/importance_sampling_ratio/max": 1.0543627738952637, "entropy": 0.03847737517207861, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.99944007396698, "reward_meter_mean": 0.99944007396698, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.99944007396698, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:37:03Z", "mode": "train", "global_step": 2728, "epoch": 0.10957143430935454, "loss": -0.0004, "grad_norm": 1.235407829284668, "learning_rate": 1.7363636363636366e-06, "num_tokens": 6178952.0, "completions/mean_length": 32.0, "completions/min_length": 32.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9992912411689758, "rewards/meter/std": 1.968258584383875e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992912411689758, "rewards/total_composite/std": 1.968258584383875e-05, "reward": 0.9992912411689758, "reward_std": 1.9688604879775085e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0030222726054489613, "sampling/sampling_logp_difference/max": 0.4845571517944336, "sampling/importance_sampling_ratio/min": 0.6159699559211731, "sampling/importance_sampling_ratio/mean": 0.9995523691177368, "sampling/importance_sampling_ratio/max": 1.0773866176605225, "entropy": 0.011132915853522718, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9992912411689758, "reward_meter_mean": 0.9992912411689758, "reward_meter_std": 1.968258584383875e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992912411689758, "reward_total_composite_std": 1.968258584383875e-05} {"timestamp_utc": "2026-04-12T02:37:11Z", "mode": "train", "global_step": 2729, "epoch": 0.10961159979113949, "loss": 0.0017, "grad_norm": 1.5231090784072876, "learning_rate": 1.7333333333333336e-06, "num_tokens": 6183584.0, "completions/mean_length": 345.0, "completions/min_length": 339.0, "completions/max_length": 359.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 345.0, "completions/min_terminated_length": 339.0, "completions/max_terminated_length": 359.0, "rewards/meter/mean": 0.9989666938781738, "rewards/meter/std": 0.00017414879403077066, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9558823108673096, "rewards/repeat_penalty/std": 0.0608881339430809, "rewards/total_composite/mean": 0.8594046235084534, "rewards/total_composite/std": 0.05473008379340172, "reward": 0.8594046235084534, "reward_std": 0.05473008751869202, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05097750574350357, "sampling/sampling_logp_difference/max": 1.327122688293457, "sampling/importance_sampling_ratio/min": 0.2652393579483032, "sampling/importance_sampling_ratio/mean": 1.0122843980789185, "sampling/importance_sampling_ratio/max": 1.6215828657150269, "entropy": 0.4432450830936432, "clip_ratio/low_mean": 0.01417952380143106, "clip_ratio/low_min": 0.01417952380143106, "clip_ratio/high_mean": 0.0216883085668087, "clip_ratio/high_max": 0.0216883085668087, "clip_ratio/region_mean": 0.03586783236823976, "reward_total_mean": 0.8594046235084534, "reward_meter_mean": 0.9989666938781738, "reward_meter_std": 0.00017414879403077066, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9558823108673096, "reward_repeat_penalty_std": 0.0608881339430809, "reward_total_composite_mean": 0.8594046235084534, "reward_total_composite_std": 0.05473008379340172} {"timestamp_utc": "2026-04-12T02:37:15Z", "mode": "train", "global_step": 2730, "epoch": 0.10965176527292445, "loss": -0.0005, "grad_norm": 14.159218788146973, "learning_rate": 1.7303030303030304e-06, "num_tokens": 6185301.0, "completions/mean_length": 44.625, "completions/min_length": 43.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.625, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9303584098815918, "rewards/meter/std": 0.03593900799751282, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9303584098815918, "rewards/total_composite/std": 0.03593900799751282, "reward": 0.9303584098815918, "reward_std": 0.035938993096351624, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1046677976846695, "sampling/sampling_logp_difference/max": 6.265340328216553, "sampling/importance_sampling_ratio/min": 0.0019010662799701095, "sampling/importance_sampling_ratio/mean": 1.024146556854248, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37994090653955936, "clip_ratio/low_mean": 0.017045455053448677, "clip_ratio/low_min": 0.017045455053448677, "clip_ratio/high_mean": 0.061030346201732755, "clip_ratio/high_max": 0.061030346201732755, "clip_ratio/region_mean": 0.07807580125518143, "reward_total_mean": 0.9303584098815918, "reward_meter_mean": 0.9303584098815918, "reward_meter_std": 0.03593900799751282, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9303584098815918, "reward_total_composite_std": 0.03593900799751282} {"timestamp_utc": "2026-04-12T02:37:20Z", "mode": "train", "global_step": 2731, "epoch": 0.1096919307547094, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.7272727272727275e-06, "num_tokens": 6187213.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0005151918157935143, "sampling/sampling_logp_difference/max": 0.0688324123620987, "sampling/importance_sampling_ratio/min": 0.9334831237792969, "sampling/importance_sampling_ratio/mean": 0.9999112486839294, "sampling/importance_sampling_ratio/max": 1.0230506658554077, "entropy": 0.003038628783542663, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:37:24Z", "mode": "train", "global_step": 2732, "epoch": 0.10973209623649435, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.7242424242424243e-06, "num_tokens": 6188637.0, "completions/mean_length": 29.0, "completions/min_length": 29.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9957436919212341, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957436919212341, "rewards/total_composite/std": 0.0, "reward": 0.9957436919212341, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0001443149521946907, "sampling/sampling_logp_difference/max": 0.003842935897409916, "sampling/importance_sampling_ratio/min": 0.9961644411087036, "sampling/importance_sampling_ratio/mean": 1.0000722408294678, "sampling/importance_sampling_ratio/max": 1.0033706426620483, "entropy": 0.0015087932988535613, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9957436919212341, "reward_meter_mean": 0.9957436919212341, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9957436919212341, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:37:28Z", "mode": "train", "global_step": 2733, "epoch": 0.10977226171827931, "loss": 0.0137, "grad_norm": 22.582195281982422, "learning_rate": 1.7212121212121214e-06, "num_tokens": 6190107.0, "completions/mean_length": 34.75, "completions/min_length": 34.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.75, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.945763885974884, "rewards/meter/std": 0.12533576786518097, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.945763885974884, "rewards/total_composite/std": 0.12533576786518097, "reward": 0.945763885974884, "reward_std": 0.12533576786518097, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02166467346251011, "sampling/sampling_logp_difference/max": 1.478475570678711, "sampling/importance_sampling_ratio/min": 0.2279849797487259, "sampling/importance_sampling_ratio/mean": 1.0029886960983276, "sampling/importance_sampling_ratio/max": 1.4781155586242676, "entropy": 0.12251334544271231, "clip_ratio/low_mean": 0.0034722222480922937, "clip_ratio/low_min": 0.0034722222480922937, "clip_ratio/high_mean": 0.0035714285913854837, "clip_ratio/high_max": 0.0035714285913854837, "clip_ratio/region_mean": 0.0070436508394777775, "reward_total_mean": 0.945763885974884, "reward_meter_mean": 0.945763885974884, "reward_meter_std": 0.12533576786518097, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.945763885974884, "reward_total_composite_std": 0.12533576786518097} {"timestamp_utc": "2026-04-12T02:37:35Z", "mode": "train", "global_step": 2734, "epoch": 0.10981242720006426, "loss": 0.0021, "grad_norm": 2.762172222137451, "learning_rate": 1.7181818181818182e-06, "num_tokens": 6193858.0, "completions/mean_length": 268.875, "completions/min_length": 255.0, "completions/max_length": 276.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 268.875, "completions/min_terminated_length": 255.0, "completions/max_terminated_length": 276.0, "rewards/meter/mean": 0.9990511536598206, "rewards/meter/std": 0.00018971240206155926, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.990384578704834, "rewards/repeat_penalty/std": 0.027196412906050682, "rewards/total_composite/mean": 0.9894447326660156, "rewards/total_composite/std": 0.027163289487361908, "reward": 0.9894447326660156, "reward_std": 0.02716328203678131, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04823000356554985, "sampling/sampling_logp_difference/max": 2.1149091720581055, "sampling/importance_sampling_ratio/min": 0.12064424902200699, "sampling/importance_sampling_ratio/mean": 1.0080488920211792, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43885183334350586, "clip_ratio/low_mean": 0.004182156175374985, "clip_ratio/low_min": 0.004182156175374985, "clip_ratio/high_mean": 0.033498246455565095, "clip_ratio/high_max": 0.033498246455565095, "clip_ratio/region_mean": 0.03768040263094008, "reward_total_mean": 0.9894447326660156, "reward_meter_mean": 0.9990511536598206, "reward_meter_std": 0.00018971240206155926, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.990384578704834, "reward_repeat_penalty_std": 0.027196412906050682, "reward_total_composite_mean": 0.9894447326660156, "reward_total_composite_std": 0.027163289487361908} {"timestamp_utc": "2026-04-12T02:37:40Z", "mode": "train", "global_step": 2735, "epoch": 0.10985259268184921, "loss": -0.0023, "grad_norm": 2.1266746520996094, "learning_rate": 1.7151515151515152e-06, "num_tokens": 6195785.0, "completions/mean_length": 70.875, "completions/min_length": 70.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.875, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.999404788017273, "rewards/meter/std": 0.00010114238102687523, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999404788017273, "rewards/total_composite/std": 0.00010114238102687523, "reward": 0.999404788017273, "reward_std": 0.00010115459008375183, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007956513203680515, "sampling/sampling_logp_difference/max": 0.6195621490478516, "sampling/importance_sampling_ratio/min": 0.5381800532341003, "sampling/importance_sampling_ratio/mean": 1.0012547969818115, "sampling/importance_sampling_ratio/max": 1.3098119497299194, "entropy": 0.05470029590651393, "clip_ratio/low_mean": 0.0053571430034935474, "clip_ratio/low_min": 0.0053571430034935474, "clip_ratio/high_mean": 0.0035211266949772835, "clip_ratio/high_max": 0.0035211266949772835, "clip_ratio/region_mean": 0.008878269698470831, "reward_total_mean": 0.999404788017273, "reward_meter_mean": 0.999404788017273, "reward_meter_std": 0.00010114238102687523, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.999404788017273, "reward_total_composite_std": 0.00010114238102687523} {"timestamp_utc": "2026-04-12T02:37:45Z", "mode": "train", "global_step": 2736, "epoch": 0.10989275816363417, "loss": 0.0073, "grad_norm": 2.7460949420928955, "learning_rate": 1.7121212121212123e-06, "num_tokens": 6198070.0, "completions/mean_length": 122.625, "completions/min_length": 121.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.625, "completions/min_terminated_length": 121.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.9976873397827148, "rewards/meter/std": 0.0002810863661579788, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9442275762557983, "rewards/total_composite/std": 0.07359237223863602, "reward": 0.9442275762557983, "reward_std": 0.07359235733747482, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02476387470960617, "sampling/sampling_logp_difference/max": 0.8794323205947876, "sampling/importance_sampling_ratio/min": 0.4150184392929077, "sampling/importance_sampling_ratio/mean": 0.9990551471710205, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1498414622619748, "clip_ratio/low_mean": 0.004048783564940095, "clip_ratio/low_min": 0.004048783564940095, "clip_ratio/high_mean": 0.023400326492264867, "clip_ratio/high_max": 0.023400326492264867, "clip_ratio/region_mean": 0.027449110057204962, "reward_total_mean": 0.9442275762557983, "reward_meter_mean": 0.9976873397827148, "reward_meter_std": 0.0002810863661579788, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.9442275762557983, "reward_total_composite_std": 0.07359237223863602} {"timestamp_utc": "2026-04-12T02:37:50Z", "mode": "train", "global_step": 2737, "epoch": 0.10993292364541912, "loss": -0.001, "grad_norm": 0.5398823618888855, "learning_rate": 1.7090909090909091e-06, "num_tokens": 6200301.0, "completions/mean_length": 97.875, "completions/min_length": 97.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.875, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9994151592254639, "rewards/meter/std": 1.6079073247965425e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994151592254639, "rewards/total_composite/std": 1.6079073247965425e-05, "reward": 0.9994151592254639, "reward_std": 1.6088057236629538e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0048310887068510056, "sampling/sampling_logp_difference/max": 1.8052539825439453, "sampling/importance_sampling_ratio/min": 0.16443268954753876, "sampling/importance_sampling_ratio/mean": 0.9999629259109497, "sampling/importance_sampling_ratio/max": 1.3984962701797485, "entropy": 0.015126468031667173, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0012755101779475808, "clip_ratio/high_max": 0.0012755101779475808, "clip_ratio/region_mean": 0.0012755101779475808, "reward_total_mean": 0.9994151592254639, "reward_meter_mean": 0.9994151592254639, "reward_meter_std": 1.6079073247965425e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994151592254639, "reward_total_composite_std": 1.6079073247965425e-05} {"timestamp_utc": "2026-04-12T02:37:55Z", "mode": "train", "global_step": 2738, "epoch": 0.10997308912720408, "loss": -0.0017, "grad_norm": 0.7919180393218994, "learning_rate": 1.7060606060606062e-06, "num_tokens": 6202503.0, "completions/mean_length": 97.25, "completions/min_length": 96.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.25, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9980356693267822, "rewards/meter/std": 6.821734132245183e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980356693267822, "rewards/total_composite/std": 6.821734132245183e-05, "reward": 0.9980356693267822, "reward_std": 6.822366412961856e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007202071137726307, "sampling/sampling_logp_difference/max": 0.41276001930236816, "sampling/importance_sampling_ratio/min": 0.6618210673332214, "sampling/importance_sampling_ratio/mean": 1.003057837486267, "sampling/importance_sampling_ratio/max": 1.4862256050109863, "entropy": 0.06638350430876017, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.006403850042261183, "clip_ratio/high_max": 0.006403850042261183, "clip_ratio/region_mean": 0.006403850042261183, "reward_total_mean": 0.9980356693267822, "reward_meter_mean": 0.9980356693267822, "reward_meter_std": 6.821734132245183e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9980356693267822, "reward_total_composite_std": 6.821734132245183e-05} {"timestamp_utc": "2026-04-12T02:37:59Z", "mode": "train", "global_step": 2739, "epoch": 0.11001325460898903, "loss": -0.0016, "grad_norm": 4.850316524505615, "learning_rate": 1.703030303030303e-06, "num_tokens": 6204047.0, "completions/mean_length": 44.0, "completions/min_length": 42.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.0, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.9376137256622314, "rewards/meter/std": 0.004042606335133314, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9376137256622314, "rewards/total_composite/std": 0.004042606335133314, "reward": 0.9376137256622314, "reward_std": 0.004042605869472027, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0746898204088211, "sampling/sampling_logp_difference/max": 2.011808395385742, "sampling/importance_sampling_ratio/min": 0.13374659419059753, "sampling/importance_sampling_ratio/mean": 0.9946960210800171, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.21985023096203804, "clip_ratio/low_mean": 0.04031385388225317, "clip_ratio/low_min": 0.04031385388225317, "clip_ratio/high_mean": 0.016798419412225485, "clip_ratio/high_max": 0.016798419412225485, "clip_ratio/region_mean": 0.057112273294478655, "reward_total_mean": 0.9376137256622314, "reward_meter_mean": 0.9376137256622314, "reward_meter_std": 0.004042606335133314, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9376137256622314, "reward_total_composite_std": 0.004042606335133314} {"timestamp_utc": "2026-04-12T02:38:03Z", "mode": "train", "global_step": 2740, "epoch": 0.11005342009077398, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.7000000000000002e-06, "num_tokens": 6205946.0, "completions/mean_length": 66.375, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.375, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981522560119629, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981522560119629, "rewards/total_composite/std": 0.0, "reward": 0.9981522560119629, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0036742938682436943, "sampling/sampling_logp_difference/max": 0.5825433731079102, "sampling/importance_sampling_ratio/min": 0.5584760904312134, "sampling/importance_sampling_ratio/mean": 0.9993410110473633, "sampling/importance_sampling_ratio/max": 1.19320809841156, "entropy": 0.018359567504376173, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9981522560119629, "reward_meter_mean": 0.9981522560119629, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981522560119629, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:38:08Z", "mode": "train", "global_step": 2741, "epoch": 0.11009358557255894, "loss": -0.0202, "grad_norm": 1.060721755027771, "learning_rate": 1.6969696969696973e-06, "num_tokens": 6207995.0, "completions/mean_length": 96.125, "completions/min_length": 92.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.125, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9955133199691772, "rewards/meter/std": 0.004802301991730928, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9955133199691772, "rewards/total_composite/std": 0.004802301991730928, "reward": 0.9955133199691772, "reward_std": 0.004802299663424492, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006459005642682314, "sampling/sampling_logp_difference/max": 0.5894908905029297, "sampling/importance_sampling_ratio/min": 0.5616306066513062, "sampling/importance_sampling_ratio/mean": 1.0021657943725586, "sampling/importance_sampling_ratio/max": 1.8030701875686646, "entropy": 0.07564361020922661, "clip_ratio/low_mean": 0.005348057369701564, "clip_ratio/low_min": 0.005348057369701564, "clip_ratio/high_mean": 0.0012886597542092204, "clip_ratio/high_max": 0.0012886597542092204, "clip_ratio/region_mean": 0.006636717123910785, "reward_total_mean": 0.9955133199691772, "reward_meter_mean": 0.9955133199691772, "reward_meter_std": 0.004802301991730928, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9955133199691772, "reward_total_composite_std": 0.004802301991730928} {"timestamp_utc": "2026-04-12T02:38:13Z", "mode": "train", "global_step": 2742, "epoch": 0.11013375105434389, "loss": 0.0053, "grad_norm": 3.1250481605529785, "learning_rate": 1.6939393939393941e-06, "num_tokens": 6210017.0, "completions/mean_length": 100.75, "completions/min_length": 96.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.75, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.9990947842597961, "rewards/meter/std": 0.0001982693502213806, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990947842597961, "rewards/total_composite/std": 0.0001982693502213806, "reward": 0.9990947842597961, "reward_std": 0.00019826588686555624, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.043827299028635025, "sampling/sampling_logp_difference/max": 2.5691757202148438, "sampling/importance_sampling_ratio/min": 0.07659865915775299, "sampling/importance_sampling_ratio/mean": 1.0069540739059448, "sampling/importance_sampling_ratio/max": 1.9947516918182373, "entropy": 0.329579945653677, "clip_ratio/low_mean": 0.011077517876401544, "clip_ratio/low_min": 0.011077517876401544, "clip_ratio/high_mean": 0.02870650147087872, "clip_ratio/high_max": 0.02870650147087872, "clip_ratio/region_mean": 0.039784019347280264, "reward_total_mean": 0.9990947842597961, "reward_meter_mean": 0.9990947842597961, "reward_meter_std": 0.0001982693502213806, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990947842597961, "reward_total_composite_std": 0.0001982693502213806} {"timestamp_utc": "2026-04-12T02:38:18Z", "mode": "train", "global_step": 2743, "epoch": 0.11017391653612885, "loss": 0.0, "grad_norm": 2.3616504669189453, "learning_rate": 1.6909090909090912e-06, "num_tokens": 6211811.0, "completions/mean_length": 69.25, "completions/min_length": 67.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.25, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9709841012954712, "rewards/meter/std": 0.02855425886809826, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9709841012954712, "rewards/total_composite/std": 0.02855425886809826, "reward": 0.9709841012954712, "reward_std": 0.028554245829582214, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04137599095702171, "sampling/sampling_logp_difference/max": 1.5528497695922852, "sampling/importance_sampling_ratio/min": 0.3372916579246521, "sampling/importance_sampling_ratio/mean": 1.0066807270050049, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.33481148816645145, "clip_ratio/low_mean": 0.014685500762425363, "clip_ratio/low_min": 0.014685500762425363, "clip_ratio/high_mean": 0.01622902590315789, "clip_ratio/high_max": 0.01622902590315789, "clip_ratio/region_mean": 0.030914526665583253, "reward_total_mean": 0.9709841012954712, "reward_meter_mean": 0.9709841012954712, "reward_meter_std": 0.02855425886809826, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9709841012954712, "reward_total_composite_std": 0.02855425886809826} {"timestamp_utc": "2026-04-12T02:38:26Z", "mode": "train", "global_step": 2744, "epoch": 0.1102140820179138, "loss": -0.0272, "grad_norm": 3.2804408073425293, "learning_rate": 1.687878787878788e-06, "num_tokens": 6215771.0, "completions/mean_length": 294.0, "completions/min_length": 272.0, "completions/max_length": 308.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 294.0, "completions/min_terminated_length": 272.0, "completions/max_terminated_length": 308.0, "rewards/meter/mean": 0.996207058429718, "rewards/meter/std": 0.0008007285068742931, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.04629101976752281, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9663312435150146, "rewards/repeat_penalty/std": 0.039662934839725494, "rewards/total_composite/mean": 0.9384926557540894, "rewards/total_composite/std": 0.05730018764734268, "reward": 0.9384926557540894, "reward_std": 0.057300180196762085, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04739288613200188, "sampling/sampling_logp_difference/max": 1.8937530517578125, "sampling/importance_sampling_ratio/min": 0.1505059003829956, "sampling/importance_sampling_ratio/mean": 1.002259373664856, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.35994909703731537, "clip_ratio/low_mean": 0.013101526303216815, "clip_ratio/low_min": 0.013101526303216815, "clip_ratio/high_mean": 0.026779879350215197, "clip_ratio/high_max": 0.026779879350215197, "clip_ratio/region_mean": 0.03988140565343201, "reward_total_mean": 0.9384926557540894, "reward_meter_mean": 0.996207058429718, "reward_meter_std": 0.0008007285068742931, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.04629101976752281, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9663312435150146, "reward_repeat_penalty_std": 0.039662934839725494, "reward_total_composite_mean": 0.9384926557540894, "reward_total_composite_std": 0.05730018764734268} {"timestamp_utc": "2026-04-12T02:38:36Z", "mode": "train", "global_step": 2745, "epoch": 0.11025424749969875, "loss": -0.2843, "grad_norm": 0.6794636249542236, "learning_rate": 1.684848484848485e-06, "num_tokens": 6220927.0, "completions/mean_length": 481.5, "completions/min_length": 458.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 477.14288330078125, "completions/min_terminated_length": 458.0, "completions/max_terminated_length": 497.0, "rewards/meter/mean": 0.9529277086257935, "rewards/meter/std": 0.12985190749168396, "rewards/count_adherence/mean": 0.8515625, "rewards/count_adherence/std": 0.04650149121880531, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8053357601165771, "rewards/repeat_penalty/std": 0.025981293991208076, "rewards/total_composite/mean": 0.6496540307998657, "rewards/total_composite/std": 0.07202339172363281, "reward": 0.6496540307998657, "reward_std": 0.07202339917421341, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024653667584061623, "sampling/sampling_logp_difference/max": 2.698049783706665, "sampling/importance_sampling_ratio/min": 0.06733670830726624, "sampling/importance_sampling_ratio/mean": 1.0057053565979004, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17778486385941505, "clip_ratio/low_mean": 0.0029986235313117504, "clip_ratio/low_min": 0.0029986235313117504, "clip_ratio/high_mean": 0.012085011578164995, "clip_ratio/high_max": 0.012085011578164995, "clip_ratio/region_mean": 0.015083635109476745, "reward_total_mean": 0.6496540307998657, "reward_meter_mean": 0.9529277086257935, "reward_meter_std": 0.12985190749168396, "reward_count_adherence_mean": 0.8515625, "reward_count_adherence_std": 0.04650149121880531, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8053357601165771, "reward_repeat_penalty_std": 0.025981293991208076, "reward_total_composite_mean": 0.6496540307998657, "reward_total_composite_std": 0.07202339172363281} {"timestamp_utc": "2026-04-12T02:38:40Z", "mode": "train", "global_step": 2746, "epoch": 0.11029441298148371, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.6818181818181819e-06, "num_tokens": 6222631.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0003325922298245132, "sampling/sampling_logp_difference/max": 0.06264449656009674, "sampling/importance_sampling_ratio/min": 0.9392772912979126, "sampling/importance_sampling_ratio/mean": 0.9999988079071045, "sampling/importance_sampling_ratio/max": 1.0122551918029785, "entropy": 0.002097451186273247, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:38:45Z", "mode": "train", "global_step": 2747, "epoch": 0.11033457846326866, "loss": 0.0146, "grad_norm": 3.3234918117523193, "learning_rate": 1.678787878787879e-06, "num_tokens": 6224387.0, "completions/mean_length": 68.5, "completions/min_length": 65.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.5, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.973109781742096, "rewards/meter/std": 0.031855545938014984, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.973109781742096, "rewards/total_composite/std": 0.031855545938014984, "reward": 0.973109781742096, "reward_std": 0.031855542212724686, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.038049403578042984, "sampling/sampling_logp_difference/max": 1.5401935577392578, "sampling/importance_sampling_ratio/min": 0.21433962881565094, "sampling/importance_sampling_ratio/mean": 1.0069949626922607, "sampling/importance_sampling_ratio/max": 1.755409598350525, "entropy": 0.2892153840512037, "clip_ratio/low_mean": 0.0017857142956927419, "clip_ratio/low_min": 0.0017857142956927419, "clip_ratio/high_mean": 0.020309405983425677, "clip_ratio/high_max": 0.020309405983425677, "clip_ratio/region_mean": 0.02209512027911842, "reward_total_mean": 0.973109781742096, "reward_meter_mean": 0.973109781742096, "reward_meter_std": 0.031855545938014984, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.973109781742096, "reward_total_composite_std": 0.031855545938014984} {"timestamp_utc": "2026-04-12T02:38:52Z", "mode": "train", "global_step": 2748, "epoch": 0.11037474394505362, "loss": 0.0236, "grad_norm": 1.725966215133667, "learning_rate": 1.675757575757576e-06, "num_tokens": 6228377.0, "completions/mean_length": 296.75, "completions/min_length": 268.0, "completions/max_length": 309.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 296.75, "completions/min_terminated_length": 268.0, "completions/max_terminated_length": 309.0, "rewards/meter/mean": 0.9991159439086914, "rewards/meter/std": 0.0003877849958371371, "rewards/count_adherence/mean": 0.890625, "rewards/count_adherence/std": 0.04419417306780815, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9171568751335144, "rewards/repeat_penalty/std": 0.033501796424388885, "rewards/total_composite/mean": 0.8153371810913086, "rewards/total_composite/std": 0.031080491840839386, "reward": 0.8153371810913086, "reward_std": 0.031080493703484535, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025589879602193832, "sampling/sampling_logp_difference/max": 3.3428661823272705, "sampling/importance_sampling_ratio/min": 0.03533553332090378, "sampling/importance_sampling_ratio/mean": 1.0036824941635132, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.16522958502173424, "clip_ratio/low_mean": 0.005000000121071935, "clip_ratio/low_min": 0.005000000121071935, "clip_ratio/high_mean": 0.01721590873785317, "clip_ratio/high_max": 0.01721590873785317, "clip_ratio/region_mean": 0.022215908858925104, "reward_total_mean": 0.8153371810913086, "reward_meter_mean": 0.9991159439086914, "reward_meter_std": 0.0003877849958371371, "reward_count_adherence_mean": 0.890625, "reward_count_adherence_std": 0.04419417306780815, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9171568751335144, "reward_repeat_penalty_std": 0.033501796424388885, "reward_total_composite_mean": 0.8153371810913086, "reward_total_composite_std": 0.031080491840839386} {"timestamp_utc": "2026-04-12T02:38:57Z", "mode": "train", "global_step": 2749, "epoch": 0.11041490942683857, "loss": -0.0163, "grad_norm": 3.2416765689849854, "learning_rate": 1.6727272727272728e-06, "num_tokens": 6230560.0, "completions/mean_length": 101.875, "completions/min_length": 99.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.875, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.9238500595092773, "rewards/meter/std": 0.10599032789468765, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9238500595092773, "rewards/total_composite/std": 0.10599032789468765, "reward": 0.9238500595092773, "reward_std": 0.10599032789468765, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03779112920165062, "sampling/sampling_logp_difference/max": 1.4791641235351562, "sampling/importance_sampling_ratio/min": 0.22782805562019348, "sampling/importance_sampling_ratio/mean": 1.008776307106018, "sampling/importance_sampling_ratio/max": 1.7313110828399658, "entropy": 0.36418476328253746, "clip_ratio/low_mean": 0.011289363959804177, "clip_ratio/low_min": 0.011289363959804177, "clip_ratio/high_mean": 0.019543501897715032, "clip_ratio/high_max": 0.019543501897715032, "clip_ratio/region_mean": 0.03083286585751921, "reward_total_mean": 0.9238500595092773, "reward_meter_mean": 0.9238500595092773, "reward_meter_std": 0.10599032789468765, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9238500595092773, "reward_total_composite_std": 0.10599032789468765} {"timestamp_utc": "2026-04-12T02:39:02Z", "mode": "train", "global_step": 2750, "epoch": 0.11045507490862352, "loss": -0.0042, "grad_norm": 3.4270999431610107, "learning_rate": 1.6696969696969698e-06, "num_tokens": 6232648.0, "completions/mean_length": 97.0, "completions/min_length": 96.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.0, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9976779818534851, "rewards/meter/std": 0.0009901836747303605, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9976779818534851, "rewards/total_composite/std": 0.0009901836747303605, "reward": 0.9976779818534851, "reward_std": 0.0009901747107505798, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005543787498027086, "sampling/sampling_logp_difference/max": 0.7107429504394531, "sampling/importance_sampling_ratio/min": 0.49127906560897827, "sampling/importance_sampling_ratio/mean": 1.0021347999572754, "sampling/importance_sampling_ratio/max": 1.6211403608322144, "entropy": 0.07637950498610735, "clip_ratio/low_mean": 0.0013020833721384406, "clip_ratio/low_min": 0.0013020833721384406, "clip_ratio/high_mean": 0.008981169667094946, "clip_ratio/high_max": 0.008981169667094946, "clip_ratio/region_mean": 0.010283253039233387, "reward_total_mean": 0.9976779818534851, "reward_meter_mean": 0.9976779818534851, "reward_meter_std": 0.0009901836747303605, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9976779818534851, "reward_total_composite_std": 0.0009901836747303605} {"timestamp_utc": "2026-04-12T02:40:19Z", "mode": "eval", "global_step": 2750, "epoch": 0.11045507490862352, "eval_loss": NaN, "eval_runtime": 76.3731, "eval_samples_per_second": 1.362, "eval_steps_per_second": 0.17, "eval_num_tokens": 6232648.0, "eval_completions/mean_length": 208.81730769230768, "eval_completions/min_length": 59.30769230769231, "eval_completions/max_length": 410.2307692307692, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 208.81730769230768, "eval_completions/min_terminated_length": 59.30769230769231, "eval_completions/max_terminated_length": 410.2307692307692, "eval_rewards/meter/mean": 0.8095979323753943, "eval_rewards/meter/std": 0.2978983329465756, "eval_rewards/count_adherence/mean": 0.9510142528093778, "eval_rewards/count_adherence/std": 0.06824518969425789, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/repeat_penalty/mean": 0.9246861888812139, "eval_rewards/repeat_penalty/std": 0.09499467995304328, "eval_rewards/total_composite/mean": 0.7168137293595535, "eval_rewards/total_composite/std": 0.2932796369378383, "eval_reward": 0.7168137293595535, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.03025670349597931, "eval_sampling/sampling_logp_difference/max": 1.3938810641948993, "eval_sampling/importance_sampling_ratio/min": 0.26402970059559894, "eval_sampling/importance_sampling_ratio/mean": 1.0077975529890795, "eval_sampling/importance_sampling_ratio/max": 1.4980517717508168, "eval_entropy": 0.32508696615695953, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7168137293595535, "eval_reward_meter_mean": 0.8095979323753943, "eval_reward_meter_std": 0.2978983329465756, "eval_reward_count_adherence_mean": 0.9510142528093778, "eval_reward_count_adherence_std": 0.06824518969425789, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_repeat_penalty_mean": 0.9246861888812139, "eval_reward_repeat_penalty_std": 0.09499467995304328, "eval_reward_total_composite_mean": 0.7168137293595535, "eval_reward_total_composite_std": 0.2932796369378383} {"timestamp_utc": "2026-04-12T02:40:29Z", "mode": "train", "global_step": 2751, "epoch": 0.11049524039040848, "loss": -0.004, "grad_norm": 1.9583728313446045, "learning_rate": 1.6666666666666667e-06, "num_tokens": 6236726.0, "completions/mean_length": 270.75, "completions/min_length": 266.0, "completions/max_length": 276.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 270.75, "completions/min_terminated_length": 266.0, "completions/max_terminated_length": 276.0, "rewards/meter/mean": 0.9987819194793701, "rewards/meter/std": 0.0005108492914587259, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9987819194793701, "rewards/total_composite/std": 0.0005108492914587259, "reward": 0.9987819194793701, "reward_std": 0.0005108449840918183, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05048610270023346, "sampling/sampling_logp_difference/max": 1.3633785247802734, "sampling/importance_sampling_ratio/min": 0.25579512119293213, "sampling/importance_sampling_ratio/mean": 1.0091570615768433, "sampling/importance_sampling_ratio/max": 1.7512770891189575, "entropy": 0.44462598487734795, "clip_ratio/low_mean": 0.010291178710758686, "clip_ratio/low_min": 0.010291178710758686, "clip_ratio/high_mean": 0.029523379867896438, "clip_ratio/high_max": 0.029523379867896438, "clip_ratio/region_mean": 0.039814558578655124, "reward_total_mean": 0.9987819194793701, "reward_meter_mean": 0.9987819194793701, "reward_meter_std": 0.0005108492914587259, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9987819194793701, "reward_total_composite_std": 0.0005108492914587259} {"timestamp_utc": "2026-04-12T02:40:38Z", "mode": "train", "global_step": 2752, "epoch": 0.11053540587219343, "loss": 0.0122, "grad_norm": 2.32265567779541, "learning_rate": 1.6636363636363637e-06, "num_tokens": 6240441.0, "completions/mean_length": 237.375, "completions/min_length": 214.0, "completions/max_length": 252.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 237.375, "completions/min_terminated_length": 214.0, "completions/max_terminated_length": 252.0, "rewards/meter/mean": 0.9126553535461426, "rewards/meter/std": 0.10828904807567596, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8564560413360596, "rewards/repeat_penalty/std": 0.10748697072267532, "rewards/total_composite/mean": 0.7755623459815979, "rewards/total_composite/std": 0.09196323156356812, "reward": 0.7755623459815979, "reward_std": 0.09196322411298752, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04030514135956764, "sampling/sampling_logp_difference/max": 2.249701738357544, "sampling/importance_sampling_ratio/min": 0.10543066263198853, "sampling/importance_sampling_ratio/mean": 1.007071614265442, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.41799643263220787, "clip_ratio/low_mean": 0.01261913578491658, "clip_ratio/low_min": 0.01261913578491658, "clip_ratio/high_mean": 0.011767410207539797, "clip_ratio/high_max": 0.011767410207539797, "clip_ratio/region_mean": 0.024386545992456377, "reward_total_mean": 0.7755623459815979, "reward_meter_mean": 0.9126553535461426, "reward_meter_std": 0.10828904807567596, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8564560413360596, "reward_repeat_penalty_std": 0.10748697072267532, "reward_total_composite_mean": 0.7755623459815979, "reward_total_composite_std": 0.09196323156356812} {"timestamp_utc": "2026-04-12T02:40:43Z", "mode": "train", "global_step": 2753, "epoch": 0.11057557135397839, "loss": 0.0132, "grad_norm": 12.199896812438965, "learning_rate": 1.6606060606060605e-06, "num_tokens": 6241917.0, "completions/mean_length": 44.5, "completions/min_length": 41.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.5, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.930524468421936, "rewards/meter/std": 0.031606100499629974, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.930524468421936, "rewards/total_composite/std": 0.031606100499629974, "reward": 0.930524468421936, "reward_std": 0.03160610795021057, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07848232239484787, "sampling/sampling_logp_difference/max": 2.279867649078369, "sampling/importance_sampling_ratio/min": 0.10229774564504623, "sampling/importance_sampling_ratio/mean": 0.9975581765174866, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2707332409918308, "clip_ratio/low_mean": 0.0029761905316263437, "clip_ratio/low_min": 0.0029761905316263437, "clip_ratio/high_mean": 0.05645520123653114, "clip_ratio/high_max": 0.05645520123653114, "clip_ratio/region_mean": 0.05943139176815748, "reward_total_mean": 0.930524468421936, "reward_meter_mean": 0.930524468421936, "reward_meter_std": 0.031606100499629974, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.930524468421936, "reward_total_composite_std": 0.031606100499629974} {"timestamp_utc": "2026-04-12T02:40:54Z", "mode": "train", "global_step": 2754, "epoch": 0.11061573683576334, "loss": 0.0001, "grad_norm": 1.868078351020813, "learning_rate": 1.6575757575757578e-06, "num_tokens": 6246685.0, "completions/mean_length": 400.0, "completions/min_length": 363.0, "completions/max_length": 450.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 400.0, "completions/min_terminated_length": 363.0, "completions/max_terminated_length": 450.0, "rewards/meter/mean": 0.9454325437545776, "rewards/meter/std": 0.0641726702451706, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.033064987510442734, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8041722774505615, "rewards/repeat_penalty/std": 0.09782867133617401, "rewards/total_composite/mean": 0.667221188545227, "rewards/total_composite/std": 0.10969909280538559, "reward": 0.667221188545227, "reward_std": 0.10969908535480499, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.048431482166051865, "sampling/sampling_logp_difference/max": 10.797205924987793, "sampling/importance_sampling_ratio/min": 2.045658038696274e-05, "sampling/importance_sampling_ratio/mean": 1.0145169496536255, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45330196619033813, "clip_ratio/low_mean": 0.017272761091589928, "clip_ratio/low_min": 0.017272761091589928, "clip_ratio/high_mean": 0.016477052122354507, "clip_ratio/high_max": 0.016477052122354507, "clip_ratio/region_mean": 0.033749813213944435, "reward_total_mean": 0.667221188545227, "reward_meter_mean": 0.9454325437545776, "reward_meter_std": 0.0641726702451706, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.033064987510442734, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8041722774505615, "reward_repeat_penalty_std": 0.09782867133617401, "reward_total_composite_mean": 0.667221188545227, "reward_total_composite_std": 0.10969909280538559} {"timestamp_utc": "2026-04-12T02:41:00Z", "mode": "train", "global_step": 2755, "epoch": 0.1106559023175483, "loss": -0.0026, "grad_norm": 0.17323201894760132, "learning_rate": 1.6545454545454548e-06, "num_tokens": 6248505.0, "completions/mean_length": 71.5, "completions/min_length": 71.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.5, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9994492530822754, "rewards/meter/std": 2.2336023903335445e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994492530822754, "rewards/total_composite/std": 2.2336023903335445e-05, "reward": 0.9994492530822754, "reward_std": 2.232198130514007e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0045864214189350605, "sampling/sampling_logp_difference/max": 0.5479953289031982, "sampling/importance_sampling_ratio/min": 0.5781075954437256, "sampling/importance_sampling_ratio/mean": 1.0008584260940552, "sampling/importance_sampling_ratio/max": 1.0643366575241089, "entropy": 0.031015932094305754, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0016666667070239782, "clip_ratio/high_max": 0.0016666667070239782, "clip_ratio/region_mean": 0.0016666667070239782, "reward_total_mean": 0.9994492530822754, "reward_meter_mean": 0.9994492530822754, "reward_meter_std": 2.2336023903335445e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994492530822754, "reward_total_composite_std": 2.2336023903335445e-05} {"timestamp_utc": "2026-04-12T02:41:07Z", "mode": "train", "global_step": 2756, "epoch": 0.11069606779933325, "loss": -0.0331, "grad_norm": 2.2315144538879395, "learning_rate": 1.6515151515151517e-06, "num_tokens": 6251781.0, "completions/mean_length": 204.5, "completions/min_length": 181.0, "completions/max_length": 213.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 204.5, "completions/min_terminated_length": 181.0, "completions/max_terminated_length": 213.0, "rewards/meter/mean": 0.9686325192451477, "rewards/meter/std": 0.029917554929852486, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8901515007019043, "rewards/repeat_penalty/std": 0.10736672580242157, "rewards/total_composite/mean": 0.8484304547309875, "rewards/total_composite/std": 0.13986076414585114, "reward": 0.8484304547309875, "reward_std": 0.13986073434352875, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04765074700117111, "sampling/sampling_logp_difference/max": 2.9453227519989014, "sampling/importance_sampling_ratio/min": 0.05258508399128914, "sampling/importance_sampling_ratio/mean": 1.0120221376419067, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3736105412244797, "clip_ratio/low_mean": 0.009910162014421076, "clip_ratio/low_min": 0.009910162014421076, "clip_ratio/high_mean": 0.02566080493852496, "clip_ratio/high_max": 0.02566080493852496, "clip_ratio/region_mean": 0.03557096695294604, "reward_total_mean": 0.8484304547309875, "reward_meter_mean": 0.9686325192451477, "reward_meter_std": 0.029917554929852486, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8901515007019043, "reward_repeat_penalty_std": 0.10736672580242157, "reward_total_composite_mean": 0.8484304547309875, "reward_total_composite_std": 0.13986076414585114} {"timestamp_utc": "2026-04-12T02:41:15Z", "mode": "train", "global_step": 2757, "epoch": 0.1107362332811182, "loss": 0.0142, "grad_norm": 1.6318613290786743, "learning_rate": 1.6484848484848487e-06, "num_tokens": 6256090.0, "completions/mean_length": 332.625, "completions/min_length": 322.0, "completions/max_length": 336.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 332.625, "completions/min_terminated_length": 322.0, "completions/max_terminated_length": 336.0, "rewards/meter/mean": 0.9958105087280273, "rewards/meter/std": 0.006650847382843494, "rewards/count_adherence/mean": 0.9090909361839294, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9144736528396606, "rewards/repeat_penalty/std": 0.027239451184868813, "rewards/total_composite/mean": 0.8277831077575684, "rewards/total_composite/std": 0.02233772911131382, "reward": 0.8277831077575684, "reward_std": 0.022337697446346283, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02632460929453373, "sampling/sampling_logp_difference/max": 2.554610252380371, "sampling/importance_sampling_ratio/min": 0.07772251218557358, "sampling/importance_sampling_ratio/mean": 1.004244089126587, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.19466014206409454, "clip_ratio/low_mean": 0.006716518080793321, "clip_ratio/low_min": 0.006716518080793321, "clip_ratio/high_mean": 0.010339219472371042, "clip_ratio/high_max": 0.010339219472371042, "clip_ratio/region_mean": 0.017055737553164363, "reward_total_mean": 0.8277831077575684, "reward_meter_mean": 0.9958105087280273, "reward_meter_std": 0.006650847382843494, "reward_count_adherence_mean": 0.9090909361839294, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9144736528396606, "reward_repeat_penalty_std": 0.027239451184868813, "reward_total_composite_mean": 0.8277831077575684, "reward_total_composite_std": 0.02233772911131382} {"timestamp_utc": "2026-04-12T02:41:21Z", "mode": "train", "global_step": 2758, "epoch": 0.11077639876290316, "loss": 0.0014, "grad_norm": 2.0312271118164062, "learning_rate": 1.6454545454545455e-06, "num_tokens": 6257914.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9973218441009521, "rewards/meter/std": 5.1264036301290616e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973218441009521, "rewards/total_composite/std": 5.1264036301290616e-05, "reward": 0.9973218441009521, "reward_std": 5.1269918913021684e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008693864569067955, "sampling/sampling_logp_difference/max": 0.5072799921035767, "sampling/importance_sampling_ratio/min": 0.6021311283111572, "sampling/importance_sampling_ratio/mean": 0.9973583221435547, "sampling/importance_sampling_ratio/max": 1.3268663883209229, "entropy": 0.04664692562073469, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.010245901066809893, "clip_ratio/high_max": 0.010245901066809893, "clip_ratio/region_mean": 0.010245901066809893, "reward_total_mean": 0.9973218441009521, "reward_meter_mean": 0.9973218441009521, "reward_meter_std": 5.1264036301290616e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9973218441009521, "reward_total_composite_std": 5.1264036301290616e-05} {"timestamp_utc": "2026-04-12T02:41:29Z", "mode": "train", "global_step": 2759, "epoch": 0.11081656424468811, "loss": -0.0188, "grad_norm": 1.848800539970398, "learning_rate": 1.6424242424242426e-06, "num_tokens": 6262570.0, "completions/mean_length": 345.0, "completions/min_length": 321.0, "completions/max_length": 353.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 345.0, "completions/min_terminated_length": 321.0, "completions/max_terminated_length": 353.0, "rewards/meter/mean": 0.9986103773117065, "rewards/meter/std": 0.000737982802093029, "rewards/count_adherence/mean": 0.9861111044883728, "rewards/count_adherence/std": 0.03928370773792267, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9926470518112183, "rewards/repeat_penalty/std": 0.020797256380319595, "rewards/total_composite/mean": 0.9773909449577332, "rewards/total_composite/std": 0.04150521755218506, "reward": 0.9773909449577332, "reward_std": 0.04150520637631416, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04865448549389839, "sampling/sampling_logp_difference/max": 1.8600101470947266, "sampling/importance_sampling_ratio/min": 0.15567104518413544, "sampling/importance_sampling_ratio/mean": 1.0085657835006714, "sampling/importance_sampling_ratio/max": 1.8393433094024658, "entropy": 0.4418103024363518, "clip_ratio/low_mean": 0.006415694952011108, "clip_ratio/low_min": 0.006415694952011108, "clip_ratio/high_mean": 0.025451545137912035, "clip_ratio/high_max": 0.025451545137912035, "clip_ratio/region_mean": 0.03186724008992314, "reward_total_mean": 0.9773909449577332, "reward_meter_mean": 0.9986103773117065, "reward_meter_std": 0.000737982802093029, "reward_count_adherence_mean": 0.9861111044883728, "reward_count_adherence_std": 0.03928370773792267, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9926470518112183, "reward_repeat_penalty_std": 0.020797256380319595, "reward_total_composite_mean": 0.9773909449577332, "reward_total_composite_std": 0.04150521755218506} {"timestamp_utc": "2026-04-12T02:41:34Z", "mode": "train", "global_step": 2760, "epoch": 0.11085672972647306, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.6393939393939396e-06, "num_tokens": 6263986.0, "completions/mean_length": 35.0, "completions/min_length": 35.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9980231523513794, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980231523513794, "rewards/total_composite/std": 0.0, "reward": 0.9980231523513794, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00032912864116951823, "sampling/sampling_logp_difference/max": 0.0035952008329331875, "sampling/importance_sampling_ratio/min": 0.9977269768714905, "sampling/importance_sampling_ratio/mean": 1.000309944152832, "sampling/importance_sampling_ratio/max": 1.0036016702651978, "entropy": 0.0025047636299859732, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9980231523513794, "reward_meter_mean": 0.9980231523513794, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9980231523513794, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:41:42Z", "mode": "train", "global_step": 2761, "epoch": 0.11089689520825802, "loss": 0.0416, "grad_norm": 2.0906882286071777, "learning_rate": 1.6363636363636365e-06, "num_tokens": 6268093.0, "completions/mean_length": 319.375, "completions/min_length": 306.0, "completions/max_length": 342.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 319.375, "completions/min_terminated_length": 306.0, "completions/max_terminated_length": 342.0, "rewards/meter/mean": 0.9989085793495178, "rewards/meter/std": 0.00021208541875239462, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0578637570142746, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9769607782363892, "rewards/repeat_penalty/std": 0.03188912943005562, "rewards/total_composite/mean": 0.946518063545227, "rewards/total_composite/std": 0.07988038659095764, "reward": 0.946518063545227, "reward_std": 0.07988038659095764, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.050234466791152954, "sampling/sampling_logp_difference/max": 1.6290464401245117, "sampling/importance_sampling_ratio/min": 0.19611649215221405, "sampling/importance_sampling_ratio/mean": 1.0140368938446045, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47739846259355545, "clip_ratio/low_mean": 0.007310603512451053, "clip_ratio/low_min": 0.007310603512451053, "clip_ratio/high_mean": 0.026291712652891874, "clip_ratio/high_max": 0.026291712652891874, "clip_ratio/region_mean": 0.03360231616534293, "reward_total_mean": 0.946518063545227, "reward_meter_mean": 0.9989085793495178, "reward_meter_std": 0.00021208541875239462, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0578637570142746, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9769607782363892, "reward_repeat_penalty_std": 0.03188912943005562, "reward_total_composite_mean": 0.946518063545227, "reward_total_composite_std": 0.07988038659095764} {"timestamp_utc": "2026-04-12T02:41:47Z", "mode": "train", "global_step": 2762, "epoch": 0.11093706069004297, "loss": 0.0269, "grad_norm": 7.329155921936035, "learning_rate": 1.6333333333333335e-06, "num_tokens": 6269931.0, "completions/mean_length": 67.75, "completions/min_length": 66.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.75, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.977131724357605, "rewards/meter/std": 0.06190915405750275, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.977131724357605, "rewards/total_composite/std": 0.06190915405750275, "reward": 0.977131724357605, "reward_std": 0.06190915033221245, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04112435504794121, "sampling/sampling_logp_difference/max": 1.3632383346557617, "sampling/importance_sampling_ratio/min": 0.2558309733867645, "sampling/importance_sampling_ratio/mean": 0.9983816742897034, "sampling/importance_sampling_ratio/max": 1.5230276584625244, "entropy": 0.2622176371514797, "clip_ratio/low_mean": 0.0017857142956927419, "clip_ratio/low_min": 0.0017857142956927419, "clip_ratio/high_mean": 0.02216856903396547, "clip_ratio/high_max": 0.02216856903396547, "clip_ratio/region_mean": 0.02395428332965821, "reward_total_mean": 0.977131724357605, "reward_meter_mean": 0.977131724357605, "reward_meter_std": 0.06190915405750275, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.977131724357605, "reward_total_composite_std": 0.06190915405750275} {"timestamp_utc": "2026-04-12T02:41:52Z", "mode": "train", "global_step": 2763, "epoch": 0.11097722617182793, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.6303030303030303e-06, "num_tokens": 6271724.0, "completions/mean_length": 66.125, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.125, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981522560119629, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981522560119629, "rewards/total_composite/std": 0.0, "reward": 0.9981522560119629, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.002258425345644355, "sampling/sampling_logp_difference/max": 0.5159242153167725, "sampling/importance_sampling_ratio/min": 0.9706259965896606, "sampling/importance_sampling_ratio/mean": 1.0024110078811646, "sampling/importance_sampling_ratio/max": 1.6751859188079834, "entropy": 0.014110149233601987, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9981522560119629, "reward_meter_mean": 0.9981522560119629, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981522560119629, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:41:57Z", "mode": "train", "global_step": 2764, "epoch": 0.11101739165361288, "loss": -0.0057, "grad_norm": 3.616687774658203, "learning_rate": 1.6272727272727274e-06, "num_tokens": 6273656.0, "completions/mean_length": 79.5, "completions/min_length": 78.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.5, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9986147880554199, "rewards/meter/std": 0.0008511117775924504, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9986147880554199, "rewards/total_composite/std": 0.0008511117775924504, "reward": 0.9986147880554199, "reward_std": 0.0008511117775924504, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.029622886329889297, "sampling/sampling_logp_difference/max": 1.190237045288086, "sampling/importance_sampling_ratio/min": 0.30414915084838867, "sampling/importance_sampling_ratio/mean": 1.0039986371994019, "sampling/importance_sampling_ratio/max": 1.7587852478027344, "entropy": 0.20842239819467068, "clip_ratio/low_mean": 0.00474833813495934, "clip_ratio/low_min": 0.00474833813495934, "clip_ratio/high_mean": 0.00474683556240052, "clip_ratio/high_max": 0.00474683556240052, "clip_ratio/region_mean": 0.00949517369735986, "reward_total_mean": 0.9986147880554199, "reward_meter_mean": 0.9986147880554199, "reward_meter_std": 0.0008511117775924504, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9986147880554199, "reward_total_composite_std": 0.0008511117775924504} {"timestamp_utc": "2026-04-12T02:42:02Z", "mode": "train", "global_step": 2765, "epoch": 0.11105755713539783, "loss": -0.0034, "grad_norm": 6.583384037017822, "learning_rate": 1.6242424242424242e-06, "num_tokens": 6275778.0, "completions/mean_length": 102.25, "completions/min_length": 99.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.25, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.9677326679229736, "rewards/meter/std": 0.08840977400541306, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9677326679229736, "rewards/total_composite/std": 0.08840977400541306, "reward": 0.9677326679229736, "reward_std": 0.08840975910425186, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05027589201927185, "sampling/sampling_logp_difference/max": 1.538498878479004, "sampling/importance_sampling_ratio/min": 0.21470315754413605, "sampling/importance_sampling_ratio/mean": 1.0036083459854126, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3880934603512287, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/high_mean": 0.03991973074153066, "clip_ratio/high_max": 0.03991973074153066, "clip_ratio/region_mean": 0.04370760964229703, "reward_total_mean": 0.9677326679229736, "reward_meter_mean": 0.9677326679229736, "reward_meter_std": 0.08840977400541306, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9677326679229736, "reward_total_composite_std": 0.08840977400541306} {"timestamp_utc": "2026-04-12T02:42:08Z", "mode": "train", "global_step": 2766, "epoch": 0.11109772261718279, "loss": -0.0051, "grad_norm": 2.6680572032928467, "learning_rate": 1.6212121212121213e-06, "num_tokens": 6278065.0, "completions/mean_length": 116.875, "completions/min_length": 115.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.875, "completions/min_terminated_length": 115.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.9987287521362305, "rewards/meter/std": 0.0006380022969096899, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9987287521362305, "rewards/total_composite/std": 0.0006380022969096899, "reward": 0.9987287521362305, "reward_std": 0.0006380106206052005, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02768939547240734, "sampling/sampling_logp_difference/max": 0.7849941253662109, "sampling/importance_sampling_ratio/min": 0.45612236857414246, "sampling/importance_sampling_ratio/mean": 1.0104761123657227, "sampling/importance_sampling_ratio/max": 1.6575981378555298, "entropy": 0.2904528006911278, "clip_ratio/low_mean": 0.008585114032030106, "clip_ratio/low_min": 0.008585114032030106, "clip_ratio/high_mean": 0.013889989699237049, "clip_ratio/high_max": 0.013889989699237049, "clip_ratio/region_mean": 0.022475103731267154, "reward_total_mean": 0.9987287521362305, "reward_meter_mean": 0.9987287521362305, "reward_meter_std": 0.0006380022969096899, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9987287521362305, "reward_total_composite_std": 0.0006380022969096899} {"timestamp_utc": "2026-04-12T02:42:15Z", "mode": "train", "global_step": 2767, "epoch": 0.11113788809896774, "loss": 0.004, "grad_norm": 1.0399538278579712, "learning_rate": 1.618181818181818e-06, "num_tokens": 6281566.0, "completions/mean_length": 246.625, "completions/min_length": 244.0, "completions/max_length": 248.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 246.625, "completions/min_terminated_length": 244.0, "completions/max_terminated_length": 248.0, "rewards/meter/mean": 0.9990943074226379, "rewards/meter/std": 0.0001148659794125706, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7403846383094788, "rewards/repeat_penalty/std": 0.05723259598016739, "rewards/total_composite/mean": 0.7397112250328064, "rewards/total_composite/std": 0.05713575705885887, "reward": 0.7397112250328064, "reward_std": 0.05713575705885887, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014130277559161186, "sampling/sampling_logp_difference/max": 1.4010510444641113, "sampling/importance_sampling_ratio/min": 0.24633793532848358, "sampling/importance_sampling_ratio/mean": 1.0045311450958252, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09486087318509817, "clip_ratio/low_mean": 0.0040465755155310035, "clip_ratio/low_min": 0.0040465755155310035, "clip_ratio/high_mean": 0.005081536248326302, "clip_ratio/high_max": 0.005081536248326302, "clip_ratio/region_mean": 0.009128111763857305, "reward_total_mean": 0.7397112250328064, "reward_meter_mean": 0.9990943074226379, "reward_meter_std": 0.0001148659794125706, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7403846383094788, "reward_repeat_penalty_std": 0.05723259598016739, "reward_total_composite_mean": 0.7397112250328064, "reward_total_composite_std": 0.05713575705885887} {"timestamp_utc": "2026-04-12T02:42:20Z", "mode": "train", "global_step": 2768, "epoch": 0.1111780535807527, "loss": 0.0045, "grad_norm": 2.360358476638794, "learning_rate": 1.6151515151515153e-06, "num_tokens": 6283917.0, "completions/mean_length": 141.875, "completions/min_length": 141.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 141.875, "completions/min_terminated_length": 141.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.9991145133972168, "rewards/meter/std": 0.00046132790157571435, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8928571939468384, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.8920483589172363, "rewards/total_composite/std": 0.06576977670192719, "reward": 0.8920483589172363, "reward_std": 0.06576978415250778, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012811451219022274, "sampling/sampling_logp_difference/max": 0.977135419845581, "sampling/importance_sampling_ratio/min": 0.3763877749443054, "sampling/importance_sampling_ratio/mean": 1.0041320323944092, "sampling/importance_sampling_ratio/max": 1.6703215837478638, "entropy": 0.08987067546695471, "clip_ratio/low_mean": 0.002622719621285796, "clip_ratio/low_min": 0.002622719621285796, "clip_ratio/high_mean": 0.004426380852237344, "clip_ratio/high_max": 0.004426380852237344, "clip_ratio/region_mean": 0.00704910047352314, "reward_total_mean": 0.8920483589172363, "reward_meter_mean": 0.9991145133972168, "reward_meter_std": 0.00046132790157571435, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8928571939468384, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.8920483589172363, "reward_total_composite_std": 0.06576977670192719} {"timestamp_utc": "2026-04-12T02:42:30Z", "mode": "train", "global_step": 2769, "epoch": 0.11121821906253765, "loss": 0.0647, "grad_norm": 1.8239376544952393, "learning_rate": 1.6121212121212124e-06, "num_tokens": 6288893.0, "completions/mean_length": 501.0, "completions/min_length": 475.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 497.3333435058594, "completions/min_terminated_length": 475.0, "completions/max_terminated_length": 506.0, "rewards/meter/mean": 0.9878989458084106, "rewards/meter/std": 0.03154898062348366, "rewards/count_adherence/mean": 0.7573529481887817, "rewards/count_adherence/std": 0.020797256380319595, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.97975754737854, "rewards/repeat_penalty/std": 0.021684719249606133, "rewards/total_composite/mean": 0.7339591979980469, "rewards/total_composite/std": 0.05060318857431412, "reward": 0.7339591979980469, "reward_std": 0.050603192299604416, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05508255213499069, "sampling/sampling_logp_difference/max": 2.9603583812713623, "sampling/importance_sampling_ratio/min": 0.05180035158991814, "sampling/importance_sampling_ratio/mean": 1.0112419128417969, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37871410325169563, "clip_ratio/low_mean": 0.010852963663637638, "clip_ratio/low_min": 0.010852963663637638, "clip_ratio/high_mean": 0.014446925837546587, "clip_ratio/high_max": 0.014446925837546587, "clip_ratio/region_mean": 0.025299889501184225, "reward_total_mean": 0.7339591979980469, "reward_meter_mean": 0.9878989458084106, "reward_meter_std": 0.03154898062348366, "reward_count_adherence_mean": 0.7573529481887817, "reward_count_adherence_std": 0.020797256380319595, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.97975754737854, "reward_repeat_penalty_std": 0.021684719249606133, "reward_total_composite_mean": 0.7339591979980469, "reward_total_composite_std": 0.05060318857431412} {"timestamp_utc": "2026-04-12T02:42:35Z", "mode": "train", "global_step": 2770, "epoch": 0.1112583845443226, "loss": -0.0003, "grad_norm": 0.033189766108989716, "learning_rate": 1.6090909090909092e-06, "num_tokens": 6290734.0, "completions/mean_length": 71.125, "completions/min_length": 71.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.125, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9994409084320068, "rewards/meter/std": 2.465912302795914e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994409084320068, "rewards/total_composite/std": 2.465912302795914e-06, "reward": 0.9994409084320068, "reward_std": 2.4565813419030746e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0038025446701794863, "sampling/sampling_logp_difference/max": 0.2554647922515869, "sampling/importance_sampling_ratio/min": 0.7745563983917236, "sampling/importance_sampling_ratio/mean": 1.0016711950302124, "sampling/importance_sampling_ratio/max": 1.0757405757904053, "entropy": 0.03055504336953163, "clip_ratio/low_mean": 0.0017361111240461469, "clip_ratio/low_min": 0.0017361111240461469, "clip_ratio/high_mean": 0.0017605633474886417, "clip_ratio/high_max": 0.0017605633474886417, "clip_ratio/region_mean": 0.0034966744715347886, "reward_total_mean": 0.9994409084320068, "reward_meter_mean": 0.9994409084320068, "reward_meter_std": 2.465912302795914e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994409084320068, "reward_total_composite_std": 2.465912302795914e-06} {"timestamp_utc": "2026-04-12T02:42:40Z", "mode": "train", "global_step": 2771, "epoch": 0.11129855002610756, "loss": 0.0169, "grad_norm": 4.650167465209961, "learning_rate": 1.6060606060606063e-06, "num_tokens": 6293263.0, "completions/mean_length": 135.125, "completions/min_length": 130.0, "completions/max_length": 139.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 135.125, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 139.0, "rewards/meter/mean": 0.9964702725410461, "rewards/meter/std": 0.003808342618867755, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9964702725410461, "rewards/total_composite/std": 0.003808342618867755, "reward": 0.9964702725410461, "reward_std": 0.003808349836617708, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04582271724939346, "sampling/sampling_logp_difference/max": 1.5504403114318848, "sampling/importance_sampling_ratio/min": 0.2121545374393463, "sampling/importance_sampling_ratio/mean": 1.0039819478988647, "sampling/importance_sampling_ratio/max": 1.650173544883728, "entropy": 0.4557015933096409, "clip_ratio/low_mean": 0.009322975529357791, "clip_ratio/low_min": 0.009322975529357791, "clip_ratio/high_mean": 0.023043142282404006, "clip_ratio/high_max": 0.023043142282404006, "clip_ratio/region_mean": 0.032366117811761796, "reward_total_mean": 0.9964702725410461, "reward_meter_mean": 0.9964702725410461, "reward_meter_std": 0.003808342618867755, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9964702725410461, "reward_total_composite_std": 0.003808342618867755} {"timestamp_utc": "2026-04-12T02:42:45Z", "mode": "train", "global_step": 2772, "epoch": 0.11133871550789251, "loss": 0.0239, "grad_norm": 2.930152416229248, "learning_rate": 1.6030303030303033e-06, "num_tokens": 6295009.0, "completions/mean_length": 67.25, "completions/min_length": 65.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.25, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9888796210289001, "rewards/meter/std": 0.006786399520933628, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9888796210289001, "rewards/total_composite/std": 0.006786399520933628, "reward": 0.9888796210289001, "reward_std": 0.006786399520933628, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028971223160624504, "sampling/sampling_logp_difference/max": 1.6678972244262695, "sampling/importance_sampling_ratio/min": 0.18864330649375916, "sampling/importance_sampling_ratio/mean": 1.0060391426086426, "sampling/importance_sampling_ratio/max": 1.6033055782318115, "entropy": 0.2543229255825281, "clip_ratio/low_mean": 0.008928571594879031, "clip_ratio/low_min": 0.008928571594879031, "clip_ratio/high_mean": 0.00938489381223917, "clip_ratio/high_max": 0.00938489381223917, "clip_ratio/region_mean": 0.0183134654071182, "reward_total_mean": 0.9888796210289001, "reward_meter_mean": 0.9888796210289001, "reward_meter_std": 0.006786399520933628, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9888796210289001, "reward_total_composite_std": 0.006786399520933628} {"timestamp_utc": "2026-04-12T02:42:50Z", "mode": "train", "global_step": 2773, "epoch": 0.11137888098967746, "loss": 0.0032, "grad_norm": 5.738377094268799, "learning_rate": 1.6000000000000001e-06, "num_tokens": 6296752.0, "completions/mean_length": 60.875, "completions/min_length": 60.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.875, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9969919919967651, "rewards/meter/std": 0.0007282263832166791, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9969919919967651, "rewards/total_composite/std": 0.0007282263832166791, "reward": 0.9969919919967651, "reward_std": 0.000728218408767134, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01134834811091423, "sampling/sampling_logp_difference/max": 0.8574669361114502, "sampling/importance_sampling_ratio/min": 0.6171859502792358, "sampling/importance_sampling_ratio/mean": 1.0058006048202515, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.060230674454942346, "clip_ratio/low_mean": 0.012500000186264515, "clip_ratio/low_min": 0.012500000186264515, "clip_ratio/high_mean": 0.0020491802133619785, "clip_ratio/high_max": 0.0020491802133619785, "clip_ratio/region_mean": 0.014549180399626493, "reward_total_mean": 0.9969919919967651, "reward_meter_mean": 0.9969919919967651, "reward_meter_std": 0.0007282263832166791, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9969919919967651, "reward_total_composite_std": 0.0007282263832166791} {"timestamp_utc": "2026-04-12T02:42:56Z", "mode": "train", "global_step": 2774, "epoch": 0.11141904647146242, "loss": 0.0355, "grad_norm": 2.424381971359253, "learning_rate": 1.5969696969696972e-06, "num_tokens": 6299532.0, "completions/mean_length": 173.5, "completions/min_length": 166.0, "completions/max_length": 183.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 173.5, "completions/min_terminated_length": 166.0, "completions/max_terminated_length": 183.0, "rewards/meter/mean": 0.9787227511405945, "rewards/meter/std": 0.03248432278633118, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9444444179534912, "rewards/repeat_penalty/std": 0.08399210125207901, "rewards/total_composite/mean": 0.9236654043197632, "rewards/total_composite/std": 0.08021914213895798, "reward": 0.9236654043197632, "reward_std": 0.08021914213895798, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04096399247646332, "sampling/sampling_logp_difference/max": 1.5549349784851074, "sampling/importance_sampling_ratio/min": 0.21120309829711914, "sampling/importance_sampling_ratio/mean": 1.0065571069717407, "sampling/importance_sampling_ratio/max": 1.7961452007293701, "entropy": 0.36277873255312443, "clip_ratio/low_mean": 0.013373102992773056, "clip_ratio/low_min": 0.013373102992773056, "clip_ratio/high_mean": 0.01708856108598411, "clip_ratio/high_max": 0.01708856108598411, "clip_ratio/region_mean": 0.030461664078757167, "reward_total_mean": 0.9236654043197632, "reward_meter_mean": 0.9787227511405945, "reward_meter_std": 0.03248432278633118, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9444444179534912, "reward_repeat_penalty_std": 0.08399210125207901, "reward_total_composite_mean": 0.9236654043197632, "reward_total_composite_std": 0.08021914213895798} {"timestamp_utc": "2026-04-12T02:43:00Z", "mode": "train", "global_step": 2775, "epoch": 0.11145921195324737, "loss": 0.0058, "grad_norm": 9.112265586853027, "learning_rate": 1.593939393939394e-06, "num_tokens": 6301620.0, "completions/mean_length": 86.0, "completions/min_length": 84.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.0, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.9534580707550049, "rewards/meter/std": 0.004275289364159107, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.10628911107778549, "rewards/total_composite/mean": 0.9026405811309814, "rewards/total_composite/std": 0.1037164255976677, "reward": 0.9026405811309814, "reward_std": 0.1037164255976677, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06749268621206284, "sampling/sampling_logp_difference/max": 2.1043763160705566, "sampling/importance_sampling_ratio/min": 0.12192168831825256, "sampling/importance_sampling_ratio/mean": 1.0034648180007935, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2445132229477167, "clip_ratio/low_mean": 0.013254429679363966, "clip_ratio/low_min": 0.013254429679363966, "clip_ratio/high_mean": 0.03781960904598236, "clip_ratio/high_max": 0.03781960904598236, "clip_ratio/region_mean": 0.05107403872534633, "reward_total_mean": 0.9026405811309814, "reward_meter_mean": 0.9534580707550049, "reward_meter_std": 0.004275289364159107, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.10628911107778549, "reward_total_composite_mean": 0.9026405811309814, "reward_total_composite_std": 0.1037164255976677} {"timestamp_utc": "2026-04-12T02:43:05Z", "mode": "train", "global_step": 2776, "epoch": 0.11149937743503233, "loss": 0.0002, "grad_norm": 2.9178199768066406, "learning_rate": 1.590909090909091e-06, "num_tokens": 6303505.0, "completions/mean_length": 65.625, "completions/min_length": 65.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.625, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9525059461593628, "rewards/meter/std": 0.0829981118440628, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9525059461593628, "rewards/total_composite/std": 0.0829981118440628, "reward": 0.9525059461593628, "reward_std": 0.082998126745224, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028366295620799065, "sampling/sampling_logp_difference/max": 1.2556953430175781, "sampling/importance_sampling_ratio/min": 0.2848776876926422, "sampling/importance_sampling_ratio/mean": 1.003873586654663, "sampling/importance_sampling_ratio/max": 1.5501948595046997, "entropy": 0.22676505334675312, "clip_ratio/low_mean": 0.001923076924867928, "clip_ratio/low_min": 0.001923076924867928, "clip_ratio/high_mean": 0.02643295784946531, "clip_ratio/high_max": 0.02643295784946531, "clip_ratio/region_mean": 0.02835603477433324, "reward_total_mean": 0.9525059461593628, "reward_meter_mean": 0.9525059461593628, "reward_meter_std": 0.0829981118440628, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9525059461593628, "reward_total_composite_std": 0.0829981118440628} {"timestamp_utc": "2026-04-12T02:43:10Z", "mode": "train", "global_step": 2777, "epoch": 0.1115395429168173, "loss": -0.004, "grad_norm": 2.4066951274871826, "learning_rate": 1.5878787878787879e-06, "num_tokens": 6305374.0, "completions/mean_length": 79.625, "completions/min_length": 79.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.625, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9987850189208984, "rewards/meter/std": 0.0002843133988790214, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9987850189208984, "rewards/total_composite/std": 0.0002843133988790214, "reward": 0.9987850189208984, "reward_std": 0.00028430239763110876, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02783510647714138, "sampling/sampling_logp_difference/max": 1.6474170684814453, "sampling/importance_sampling_ratio/min": 0.19254659116268158, "sampling/importance_sampling_ratio/mean": 1.0044121742248535, "sampling/importance_sampling_ratio/max": 1.57016921043396, "entropy": 0.22624119743704796, "clip_ratio/low_mean": 0.0031250000465661287, "clip_ratio/low_min": 0.0031250000465661287, "clip_ratio/high_mean": 0.010977057158015668, "clip_ratio/high_max": 0.010977057158015668, "clip_ratio/region_mean": 0.014102057204581797, "reward_total_mean": 0.9987850189208984, "reward_meter_mean": 0.9987850189208984, "reward_meter_std": 0.0002843133988790214, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9987850189208984, "reward_total_composite_std": 0.0002843133988790214} {"timestamp_utc": "2026-04-12T02:43:15Z", "mode": "train", "global_step": 2778, "epoch": 0.11157970839860225, "loss": 0.0001, "grad_norm": 2.4368531703948975, "learning_rate": 1.584848484848485e-06, "num_tokens": 6307953.0, "completions/mean_length": 132.375, "completions/min_length": 132.0, "completions/max_length": 134.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.375, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.9603220820426941, "rewards/meter/std": 0.11049286276102066, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9603220820426941, "rewards/total_composite/std": 0.11049286276102066, "reward": 0.9603220820426941, "reward_std": 0.11049285531044006, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007737660314887762, "sampling/sampling_logp_difference/max": 1.6687531471252441, "sampling/importance_sampling_ratio/min": 0.18848192691802979, "sampling/importance_sampling_ratio/mean": 1.0007283687591553, "sampling/importance_sampling_ratio/max": 1.631596565246582, "entropy": 0.051168760284781456, "clip_ratio/low_mean": 0.0009469697251915932, "clip_ratio/low_min": 0.0009469697251915932, "clip_ratio/high_mean": 0.0084732057293877, "clip_ratio/high_max": 0.0084732057293877, "clip_ratio/region_mean": 0.009420175454579294, "reward_total_mean": 0.9603220820426941, "reward_meter_mean": 0.9603220820426941, "reward_meter_std": 0.11049286276102066, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9603220820426941, "reward_total_composite_std": 0.11049286276102066} {"timestamp_utc": "2026-04-12T02:43:20Z", "mode": "train", "global_step": 2779, "epoch": 0.1116198738803872, "loss": 0.0001, "grad_norm": 2.0250144004821777, "learning_rate": 1.5818181818181818e-06, "num_tokens": 6310483.0, "completions/mean_length": 123.25, "completions/min_length": 122.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 123.25, "completions/min_terminated_length": 122.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.9977097511291504, "rewards/meter/std": 0.0001580109674250707, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977097511291504, "rewards/total_composite/std": 0.0001580109674250707, "reward": 0.9977097511291504, "reward_std": 0.00015800821711309254, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030101941898465157, "sampling/sampling_logp_difference/max": 1.549009919166565, "sampling/importance_sampling_ratio/min": 0.21245822310447693, "sampling/importance_sampling_ratio/mean": 1.0012949705123901, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.20793312788009644, "clip_ratio/low_mean": 0.013203723356127739, "clip_ratio/low_min": 0.013203723356127739, "clip_ratio/high_mean": 0.011195790022611618, "clip_ratio/high_max": 0.011195790022611618, "clip_ratio/region_mean": 0.024399513378739357, "reward_total_mean": 0.9977097511291504, "reward_meter_mean": 0.9977097511291504, "reward_meter_std": 0.0001580109674250707, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9977097511291504, "reward_total_composite_std": 0.0001580109674250707} {"timestamp_utc": "2026-04-12T02:43:28Z", "mode": "train", "global_step": 2780, "epoch": 0.11166003936217216, "loss": 0.0047, "grad_norm": 1.3197673559188843, "learning_rate": 1.5787878787878788e-06, "num_tokens": 6314718.0, "completions/mean_length": 323.375, "completions/min_length": 317.0, "completions/max_length": 355.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 323.375, "completions/min_terminated_length": 317.0, "completions/max_terminated_length": 355.0, "rewards/meter/mean": 0.9988690614700317, "rewards/meter/std": 0.00020919565577059984, "rewards/count_adherence/mean": 0.9124999642372131, "rewards/count_adherence/std": 0.0353553481400013, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8053405284881592, "rewards/repeat_penalty/std": 0.046673085540533066, "rewards/total_composite/mean": 0.7331855297088623, "rewards/total_composite/std": 0.03377552330493927, "reward": 0.7331855297088623, "reward_std": 0.033775512129068375, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0219265203922987, "sampling/sampling_logp_difference/max": 4.522955417633057, "sampling/importance_sampling_ratio/min": 0.010856889188289642, "sampling/importance_sampling_ratio/mean": 1.0070363283157349, "sampling/importance_sampling_ratio/max": 1.8094199895858765, "entropy": 0.19129730947315693, "clip_ratio/low_mean": 0.0015637245669495314, "clip_ratio/low_min": 0.0015637245669495314, "clip_ratio/high_mean": 0.010822767158970237, "clip_ratio/high_max": 0.010822767158970237, "clip_ratio/region_mean": 0.012386491725919768, "reward_total_mean": 0.7331855297088623, "reward_meter_mean": 0.9988690614700317, "reward_meter_std": 0.00020919565577059984, "reward_count_adherence_mean": 0.9124999642372131, "reward_count_adherence_std": 0.0353553481400013, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8053405284881592, "reward_repeat_penalty_std": 0.046673085540533066, "reward_total_composite_mean": 0.7331855297088623, "reward_total_composite_std": 0.03377552330493927} {"timestamp_utc": "2026-04-12T02:43:33Z", "mode": "train", "global_step": 2781, "epoch": 0.11170020484395711, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.5757575757575759e-06, "num_tokens": 6316374.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00022916619491297752, "sampling/sampling_logp_difference/max": 0.002041937317699194, "sampling/importance_sampling_ratio/min": 0.9993596076965332, "sampling/importance_sampling_ratio/mean": 1.0002243518829346, "sampling/importance_sampling_ratio/max": 1.0020439624786377, "entropy": 0.0019741212599910796, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:43:38Z", "mode": "train", "global_step": 2782, "epoch": 0.11174037032574206, "loss": -0.0038, "grad_norm": 3.458383560180664, "learning_rate": 1.572727272727273e-06, "num_tokens": 6318445.0, "completions/mean_length": 101.875, "completions/min_length": 98.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.875, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.998963475227356, "rewards/meter/std": 0.0003066870558541268, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998963475227356, "rewards/total_composite/std": 0.0003066870558541268, "reward": 0.998963475227356, "reward_std": 0.00030668076942674816, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.059917621314525604, "sampling/sampling_logp_difference/max": 3.0749526023864746, "sampling/importance_sampling_ratio/min": 0.04619181901216507, "sampling/importance_sampling_ratio/mean": 1.0039867162704468, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3706792779266834, "clip_ratio/low_mean": 0.01848576357588172, "clip_ratio/low_min": 0.01848576357588172, "clip_ratio/high_mean": 0.02071394305676222, "clip_ratio/high_max": 0.02071394305676222, "clip_ratio/region_mean": 0.03919970663264394, "reward_total_mean": 0.998963475227356, "reward_meter_mean": 0.998963475227356, "reward_meter_std": 0.0003066870558541268, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.998963475227356, "reward_total_composite_std": 0.0003066870558541268} {"timestamp_utc": "2026-04-12T02:43:42Z", "mode": "train", "global_step": 2783, "epoch": 0.11178053580752702, "loss": 0.0798, "grad_norm": 190.96791076660156, "learning_rate": 1.56969696969697e-06, "num_tokens": 6320008.0, "completions/mean_length": 42.375, "completions/min_length": 40.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.375, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.9555168151855469, "rewards/meter/std": 0.011990510858595371, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9555168151855469, "rewards/total_composite/std": 0.011990510858595371, "reward": 0.9555168151855469, "reward_std": 0.011990513652563095, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08888134360313416, "sampling/sampling_logp_difference/max": 1.91815185546875, "sampling/importance_sampling_ratio/min": 0.14687815308570862, "sampling/importance_sampling_ratio/mean": 0.9958891868591309, "sampling/importance_sampling_ratio/max": 1.9535908699035645, "entropy": 0.3651922196149826, "clip_ratio/low_mean": 0.05653973203152418, "clip_ratio/low_min": 0.05653973203152418, "clip_ratio/high_mean": 0.014478484634310007, "clip_ratio/high_max": 0.014478484634310007, "clip_ratio/region_mean": 0.07101821666583419, "reward_total_mean": 0.9555168151855469, "reward_meter_mean": 0.9555168151855469, "reward_meter_std": 0.011990510858595371, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9555168151855469, "reward_total_composite_std": 0.011990510858595371} {"timestamp_utc": "2026-04-12T02:43:47Z", "mode": "train", "global_step": 2784, "epoch": 0.11182070128931197, "loss": 0.0038, "grad_norm": 5.360074043273926, "learning_rate": 1.566666666666667e-06, "num_tokens": 6321854.0, "completions/mean_length": 60.75, "completions/min_length": 60.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.75, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9968235492706299, "rewards/meter/std": 0.0013836961006745696, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9968235492706299, "rewards/total_composite/std": 0.0013836961006745696, "reward": 0.9968235492706299, "reward_std": 0.001383687136694789, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014301088638603687, "sampling/sampling_logp_difference/max": 1.6227011680603027, "sampling/importance_sampling_ratio/min": 0.19736485183238983, "sampling/importance_sampling_ratio/mean": 1.002347707748413, "sampling/importance_sampling_ratio/max": 1.77481210231781, "entropy": 0.07807908114045858, "clip_ratio/low_mean": 0.008333333767950535, "clip_ratio/low_min": 0.008333333767950535, "clip_ratio/high_mean": 0.010314207524061203, "clip_ratio/high_max": 0.010314207524061203, "clip_ratio/region_mean": 0.018647541292011738, "reward_total_mean": 0.9968235492706299, "reward_meter_mean": 0.9968235492706299, "reward_meter_std": 0.0013836961006745696, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9968235492706299, "reward_total_composite_std": 0.0013836961006745696} {"timestamp_utc": "2026-04-12T02:43:51Z", "mode": "train", "global_step": 2785, "epoch": 0.11186086677109693, "loss": 0.0043, "grad_norm": 2.2449779510498047, "learning_rate": 1.5636363636363638e-06, "num_tokens": 6324095.0, "completions/mean_length": 92.125, "completions/min_length": 91.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.125, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.9976515173912048, "rewards/meter/std": 0.00014370458666235209, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9976515173912048, "rewards/total_composite/std": 0.00014370458666235209, "reward": 0.9976515173912048, "reward_std": 0.00014370400458574295, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02354567125439644, "sampling/sampling_logp_difference/max": 1.614687442779541, "sampling/importance_sampling_ratio/min": 0.1989528387784958, "sampling/importance_sampling_ratio/mean": 1.0068784952163696, "sampling/importance_sampling_ratio/max": 1.9402447938919067, "entropy": 0.14724664017558098, "clip_ratio/low_mean": 0.019023986998945475, "clip_ratio/low_min": 0.019023986998945475, "clip_ratio/high_mean": 0.0054351037833839655, "clip_ratio/high_max": 0.0054351037833839655, "clip_ratio/region_mean": 0.02445909078232944, "reward_total_mean": 0.9976515173912048, "reward_meter_mean": 0.9976515173912048, "reward_meter_std": 0.00014370458666235209, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9976515173912048, "reward_total_composite_std": 0.00014370458666235209} {"timestamp_utc": "2026-04-12T02:43:56Z", "mode": "train", "global_step": 2786, "epoch": 0.11190103225288188, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.5606060606060609e-06, "num_tokens": 6325535.0, "completions/mean_length": 36.0, "completions/min_length": 36.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.0, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9996045231819153, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9996045231819153, "rewards/total_composite/std": 0.0, "reward": 0.9996045231819153, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0015392457135021687, "sampling/sampling_logp_difference/max": 0.029237329959869385, "sampling/importance_sampling_ratio/min": 0.9971944093704224, "sampling/importance_sampling_ratio/mean": 1.0015150308609009, "sampling/importance_sampling_ratio/max": 1.029668927192688, "entropy": 0.01709412969648838, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9996045231819153, "reward_meter_mean": 0.9996045231819153, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9996045231819153, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:44:00Z", "mode": "train", "global_step": 2787, "epoch": 0.11194119773466683, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.5575757575757577e-06, "num_tokens": 6327007.0, "completions/mean_length": 32.0, "completions/min_length": 32.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "reward": 0.9992982149124146, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 4.562741378322244e-05, "sampling/sampling_logp_difference/max": 0.0008772250730544329, "sampling/importance_sampling_ratio/min": 0.9996870756149292, "sampling/importance_sampling_ratio/mean": 1.00003981590271, "sampling/importance_sampling_ratio/max": 1.0008776187896729, "entropy": 0.00042030583426821977, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9992982149124146, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:44:05Z", "mode": "train", "global_step": 2788, "epoch": 0.11198136321645179, "loss": 0.0103, "grad_norm": 6.924910545349121, "learning_rate": 1.5545454545454547e-06, "num_tokens": 6329047.0, "completions/mean_length": 100.0, "completions/min_length": 93.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.0, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.9535719156265259, "rewards/meter/std": 0.12491101026535034, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9535719156265259, "rewards/total_composite/std": 0.12491101026535034, "reward": 0.9535719156265259, "reward_std": 0.12491098791360855, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.041653111577034, "sampling/sampling_logp_difference/max": 1.1875276565551758, "sampling/importance_sampling_ratio/min": 0.30497434735298157, "sampling/importance_sampling_ratio/mean": 0.9999960064888, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.349405812099576, "clip_ratio/low_mean": 0.007281553465873003, "clip_ratio/low_min": 0.007281553465873003, "clip_ratio/high_mean": 0.03628689446486533, "clip_ratio/high_max": 0.03628689446486533, "clip_ratio/region_mean": 0.04356844793073833, "reward_total_mean": 0.9535719156265259, "reward_meter_mean": 0.9535719156265259, "reward_meter_std": 0.12491101026535034, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9535719156265259, "reward_total_composite_std": 0.12491101026535034} {"timestamp_utc": "2026-04-12T02:44:09Z", "mode": "train", "global_step": 2789, "epoch": 0.11202152869823674, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.5515151515151516e-06, "num_tokens": 6330791.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00020510550530161709, "sampling/sampling_logp_difference/max": 0.002508021891117096, "sampling/importance_sampling_ratio/min": 0.9999552369117737, "sampling/importance_sampling_ratio/mean": 1.0002046823501587, "sampling/importance_sampling_ratio/max": 1.0025111436843872, "entropy": 0.0018322104588150978, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:44:13Z", "mode": "train", "global_step": 2790, "epoch": 0.1120616941800217, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.5484848484848486e-06, "num_tokens": 6332279.0, "completions/mean_length": 32.0, "completions/min_length": 32.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "reward": 0.9992982149124146, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 4.29271676694043e-05, "sampling/sampling_logp_difference/max": 0.0006322840927168727, "sampling/importance_sampling_ratio/min": 0.999407947063446, "sampling/importance_sampling_ratio/mean": 1.000036358833313, "sampling/importance_sampling_ratio/max": 1.0006325244903564, "entropy": 0.00035309882514411584, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9992982149124146, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:44:18Z", "mode": "train", "global_step": 2791, "epoch": 0.11210185966180665, "loss": 0.0171, "grad_norm": 4.835381984710693, "learning_rate": 1.5454545454545454e-06, "num_tokens": 6334035.0, "completions/mean_length": 66.5, "completions/min_length": 65.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.5, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.98065185546875, "rewards/meter/std": 0.016748299822211266, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.98065185546875, "rewards/total_composite/std": 0.016748299822211266, "reward": 0.98065185546875, "reward_std": 0.016748299822211266, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.038696978241205215, "sampling/sampling_logp_difference/max": 1.3247365951538086, "sampling/importance_sampling_ratio/min": 0.265872985124588, "sampling/importance_sampling_ratio/mean": 1.0198413133621216, "sampling/importance_sampling_ratio/max": 1.7263996601104736, "entropy": 0.35312316194176674, "clip_ratio/low_mean": 0.010924369795247912, "clip_ratio/low_min": 0.010924369795247912, "clip_ratio/high_mean": 0.005740093300119042, "clip_ratio/high_max": 0.005740093300119042, "clip_ratio/region_mean": 0.016664463095366955, "reward_total_mean": 0.98065185546875, "reward_meter_mean": 0.98065185546875, "reward_meter_std": 0.016748299822211266, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.98065185546875, "reward_total_composite_std": 0.016748299822211266} {"timestamp_utc": "2026-04-12T02:44:25Z", "mode": "train", "global_step": 2792, "epoch": 0.1121420251435916, "loss": 0.007, "grad_norm": 2.5946123600006104, "learning_rate": 1.5424242424242425e-06, "num_tokens": 6337431.0, "completions/mean_length": 232.5, "completions/min_length": 231.0, "completions/max_length": 234.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 232.5, "completions/min_terminated_length": 231.0, "completions/max_terminated_length": 234.0, "rewards/meter/mean": 0.97259521484375, "rewards/meter/std": 0.05070994421839714, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9230769276618958, "rewards/repeat_penalty/std": 0.041117113083601, "rewards/total_composite/mean": 0.8965212106704712, "rewards/total_composite/std": 0.035956308245658875, "reward": 0.8965212106704712, "reward_std": 0.035956308245658875, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025431320071220398, "sampling/sampling_logp_difference/max": 2.821336507797241, "sampling/importance_sampling_ratio/min": 0.05952633172273636, "sampling/importance_sampling_ratio/mean": 1.0048270225524902, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.19751299265772104, "clip_ratio/low_mean": 0.008609326789155602, "clip_ratio/low_min": 0.008609326789155602, "clip_ratio/high_mean": 0.011832875199615955, "clip_ratio/high_max": 0.011832875199615955, "clip_ratio/region_mean": 0.020442201988771558, "reward_total_mean": 0.8965212106704712, "reward_meter_mean": 0.97259521484375, "reward_meter_std": 0.05070994421839714, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9230769276618958, "reward_repeat_penalty_std": 0.041117113083601, "reward_total_composite_mean": 0.8965212106704712, "reward_total_composite_std": 0.035956308245658875} {"timestamp_utc": "2026-04-12T02:44:30Z", "mode": "train", "global_step": 2793, "epoch": 0.11218219062537656, "loss": -0.0046, "grad_norm": 2.338566541671753, "learning_rate": 1.5393939393939395e-06, "num_tokens": 6339557.0, "completions/mean_length": 91.75, "completions/min_length": 90.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.75, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.9973961710929871, "rewards/meter/std": 0.00040277960943058133, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973961710929871, "rewards/total_composite/std": 0.00040277960943058133, "reward": 0.9973961710929871, "reward_std": 0.00040277454536408186, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018009934574365616, "sampling/sampling_logp_difference/max": 1.1022424697875977, "sampling/importance_sampling_ratio/min": 0.3321254551410675, "sampling/importance_sampling_ratio/mean": 0.9954391717910767, "sampling/importance_sampling_ratio/max": 1.4062292575836182, "entropy": 0.10903155151754618, "clip_ratio/low_mean": 0.01233343977946788, "clip_ratio/low_min": 0.01233343977946788, "clip_ratio/high_mean": 0.006764258956536651, "clip_ratio/high_max": 0.006764258956536651, "clip_ratio/region_mean": 0.01909769873600453, "reward_total_mean": 0.9973961710929871, "reward_meter_mean": 0.9973961710929871, "reward_meter_std": 0.00040277960943058133, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9973961710929871, "reward_total_composite_std": 0.00040277960943058133} {"timestamp_utc": "2026-04-12T02:44:34Z", "mode": "train", "global_step": 2794, "epoch": 0.11222235610716151, "loss": -0.0009, "grad_norm": 4.996799468994141, "learning_rate": 1.5363636363636364e-06, "num_tokens": 6341231.0, "completions/mean_length": 69.25, "completions/min_length": 67.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.25, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9884920120239258, "rewards/meter/std": 0.01208480168133974, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9884920120239258, "rewards/total_composite/std": 0.01208480168133974, "reward": 0.9884920120239258, "reward_std": 0.012084796093404293, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03242041543126106, "sampling/sampling_logp_difference/max": 1.4227056503295898, "sampling/importance_sampling_ratio/min": 0.24106091260910034, "sampling/importance_sampling_ratio/mean": 1.006773591041565, "sampling/importance_sampling_ratio/max": 1.6237882375717163, "entropy": 0.29199461452662945, "clip_ratio/low_mean": 0.007144315168261528, "clip_ratio/low_min": 0.007144315168261528, "clip_ratio/high_mean": 0.0126050099497661, "clip_ratio/high_max": 0.0126050099497661, "clip_ratio/region_mean": 0.019749325118027627, "reward_total_mean": 0.9884920120239258, "reward_meter_mean": 0.9884920120239258, "reward_meter_std": 0.01208480168133974, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9884920120239258, "reward_total_composite_std": 0.01208480168133974} {"timestamp_utc": "2026-04-12T02:44:40Z", "mode": "train", "global_step": 2795, "epoch": 0.11226252158894647, "loss": 0.0081, "grad_norm": 3.0096166133880615, "learning_rate": 1.5333333333333334e-06, "num_tokens": 6343711.0, "completions/mean_length": 126.0, "completions/min_length": 123.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 126.0, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.9772056937217712, "rewards/meter/std": 0.017515793442726135, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9772056937217712, "rewards/total_composite/std": 0.017515793442726135, "reward": 0.9772056937217712, "reward_std": 0.017515793442726135, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04109743982553482, "sampling/sampling_logp_difference/max": 1.8138017654418945, "sampling/importance_sampling_ratio/min": 0.1630331426858902, "sampling/importance_sampling_ratio/mean": 1.0098986625671387, "sampling/importance_sampling_ratio/max": 1.8909732103347778, "entropy": 0.37405042350292206, "clip_ratio/low_mean": 0.00992763601243496, "clip_ratio/low_min": 0.00992763601243496, "clip_ratio/high_mean": 0.01491319842170924, "clip_ratio/high_max": 0.01491319842170924, "clip_ratio/region_mean": 0.0248408344341442, "reward_total_mean": 0.9772056937217712, "reward_meter_mean": 0.9772056937217712, "reward_meter_std": 0.017515793442726135, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9772056937217712, "reward_total_composite_std": 0.017515793442726135} {"timestamp_utc": "2026-04-12T02:44:49Z", "mode": "train", "global_step": 2796, "epoch": 0.11230268707073142, "loss": -0.029, "grad_norm": 1.5088597536087036, "learning_rate": 1.5303030303030302e-06, "num_tokens": 6349489.0, "completions/mean_length": 451.25, "completions/min_length": 425.0, "completions/max_length": 476.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 451.25, "completions/min_terminated_length": 425.0, "completions/max_terminated_length": 476.0, "rewards/meter/mean": 0.9978024363517761, "rewards/meter/std": 0.0035957051441073418, "rewards/count_adherence/mean": 0.7666666507720947, "rewards/count_adherence/std": 0.035634830594062805, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9777432680130005, "rewards/repeat_penalty/std": 0.023832013830542564, "rewards/total_composite/mean": 0.7476283311843872, "rewards/total_composite/std": 0.03142942115664482, "reward": 0.7476283311843872, "reward_std": 0.031429413706064224, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05049360170960426, "sampling/sampling_logp_difference/max": 5.123012065887451, "sampling/importance_sampling_ratio/min": 0.005958050023764372, "sampling/importance_sampling_ratio/mean": 1.0113346576690674, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47964709997177124, "clip_ratio/low_mean": 0.015304939355701208, "clip_ratio/low_min": 0.015304939355701208, "clip_ratio/high_mean": 0.019141072407364845, "clip_ratio/high_max": 0.019141072407364845, "clip_ratio/region_mean": 0.03444601176306605, "reward_total_mean": 0.7476283311843872, "reward_meter_mean": 0.9978024363517761, "reward_meter_std": 0.0035957051441073418, "reward_count_adherence_mean": 0.7666666507720947, "reward_count_adherence_std": 0.035634830594062805, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9777432680130005, "reward_repeat_penalty_std": 0.023832013830542564, "reward_total_composite_mean": 0.7476283311843872, "reward_total_composite_std": 0.03142942115664482} {"timestamp_utc": "2026-04-12T02:44:54Z", "mode": "train", "global_step": 2797, "epoch": 0.11234285255251637, "loss": -0.0044, "grad_norm": 2.2734148502349854, "learning_rate": 1.5272727272727275e-06, "num_tokens": 6351421.0, "completions/mean_length": 66.5, "completions/min_length": 63.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.5, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9991590976715088, "rewards/meter/std": 0.0001742392487358302, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991590976715088, "rewards/total_composite/std": 0.0001742392487358302, "reward": 0.9991590976715088, "reward_std": 0.00017422223754692823, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028453955426812172, "sampling/sampling_logp_difference/max": 1.6620092391967773, "sampling/importance_sampling_ratio/min": 0.1897573173046112, "sampling/importance_sampling_ratio/mean": 1.0024892091751099, "sampling/importance_sampling_ratio/max": 1.4288572072982788, "entropy": 0.21964257210493088, "clip_ratio/low_mean": 0.009413161547854543, "clip_ratio/low_min": 0.009413161547854543, "clip_ratio/high_mean": 0.013506086892448366, "clip_ratio/high_max": 0.013506086892448366, "clip_ratio/region_mean": 0.02291924844030291, "reward_total_mean": 0.9991590976715088, "reward_meter_mean": 0.9991590976715088, "reward_meter_std": 0.0001742392487358302, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9991590976715088, "reward_total_composite_std": 0.0001742392487358302} {"timestamp_utc": "2026-04-12T02:44:59Z", "mode": "train", "global_step": 2798, "epoch": 0.11238301803430133, "loss": -0.0027, "grad_norm": 1.5468372106552124, "learning_rate": 1.5242424242424245e-06, "num_tokens": 6353888.0, "completions/mean_length": 131.375, "completions/min_length": 128.0, "completions/max_length": 134.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.375, "completions/min_terminated_length": 128.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.9992819428443909, "rewards/meter/std": 0.00023744572536088526, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992819428443909, "rewards/total_composite/std": 0.00023744572536088526, "reward": 0.9992819428443909, "reward_std": 0.00023744485224597156, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01363647636026144, "sampling/sampling_logp_difference/max": 1.2669620513916016, "sampling/importance_sampling_ratio/min": 0.28168606758117676, "sampling/importance_sampling_ratio/mean": 1.000964641571045, "sampling/importance_sampling_ratio/max": 1.7840969562530518, "entropy": 0.06868221471086144, "clip_ratio/low_mean": 0.009618075448088348, "clip_ratio/low_min": 0.009618075448088348, "clip_ratio/high_mean": 0.0047420773771591485, "clip_ratio/high_max": 0.0047420773771591485, "clip_ratio/region_mean": 0.014360152825247496, "reward_total_mean": 0.9992819428443909, "reward_meter_mean": 0.9992819428443909, "reward_meter_std": 0.00023744572536088526, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992819428443909, "reward_total_composite_std": 0.00023744572536088526} {"timestamp_utc": "2026-04-12T02:45:10Z", "mode": "train", "global_step": 2799, "epoch": 0.11242318351608628, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.5212121212121214e-06, "num_tokens": 6355832.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9562926292419434, "rewards/meter/std": 0.0762549340724945, "rewards/count_adherence/mean": 0.71875, "rewards/count_adherence/std": 0.025877464562654495, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.9028632640838623, "rewards/repeat_penalty/std": 0.0416770875453949, "rewards/total_composite/mean": 0.5374298095703125, "rewards/total_composite/std": 0.2192150205373764, "reward": 0.5374298095703125, "reward_std": 0.2192150205373764, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.5374298095703125, "reward_meter_mean": 0.9562926292419434, "reward_meter_std": 0.0762549340724945, "reward_count_adherence_mean": 0.71875, "reward_count_adherence_std": 0.025877464562654495, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.9028632640838623, "reward_repeat_penalty_std": 0.0416770875453949, "reward_total_composite_mean": 0.5374298095703125, "reward_total_composite_std": 0.2192150205373764} {"timestamp_utc": "2026-04-12T02:45:17Z", "mode": "train", "global_step": 2800, "epoch": 0.11246334899787123, "loss": 0.0229, "grad_norm": 2.547086715698242, "learning_rate": 1.5181818181818184e-06, "num_tokens": 6359592.0, "completions/mean_length": 271.0, "completions/min_length": 257.0, "completions/max_length": 292.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 271.0, "completions/min_terminated_length": 257.0, "completions/max_terminated_length": 292.0, "rewards/meter/mean": 0.9248241186141968, "rewards/meter/std": 0.17904320359230042, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8916666507720947, "rewards/repeat_penalty/std": 0.08683133870363235, "rewards/total_composite/mean": 0.8224738240242004, "rewards/total_composite/std": 0.1752014011144638, "reward": 0.8224738240242004, "reward_std": 0.1752013862133026, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04360240325331688, "sampling/sampling_logp_difference/max": 1.5698328018188477, "sampling/importance_sampling_ratio/min": 0.2080799639225006, "sampling/importance_sampling_ratio/mean": 1.0097815990447998, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4257059842348099, "clip_ratio/low_mean": 0.00767904706299305, "clip_ratio/low_min": 0.00767904706299305, "clip_ratio/high_mean": 0.027406520443037152, "clip_ratio/high_max": 0.027406520443037152, "clip_ratio/region_mean": 0.0350855675060302, "reward_total_mean": 0.8224738240242004, "reward_meter_mean": 0.9248241186141968, "reward_meter_std": 0.17904320359230042, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8916666507720947, "reward_repeat_penalty_std": 0.08683133870363235, "reward_total_composite_mean": 0.8224738240242004, "reward_total_composite_std": 0.1752014011144638} {"timestamp_utc": "2026-04-12T02:46:35Z", "mode": "eval", "global_step": 2800, "epoch": 0.11246334899787123, "eval_loss": NaN, "eval_runtime": 78.1168, "eval_samples_per_second": 1.331, "eval_steps_per_second": 0.166, "eval_num_tokens": 6359592.0, "eval_completions/mean_length": 210.54807692307693, "eval_completions/min_length": 60.69230769230769, "eval_completions/max_length": 418.84615384615387, "eval_completions/clipped_ratio": 0.028846153846153848, "eval_completions/mean_terminated_length": 200.53846271221454, "eval_completions/min_terminated_length": 60.69230769230769, "eval_completions/max_terminated_length": 381.15384615384613, "eval_rewards/meter/mean": 0.7971470860334543, "eval_rewards/meter/std": 0.31754180788993835, "eval_rewards/count_adherence/mean": 0.9559012926541842, "eval_rewards/count_adherence/std": 0.06501825956197885, "eval_rewards/arabic_clean/mean": 0.9807692307692307, "eval_rewards/arabic_clean/std": 0.05439282839114849, "eval_rewards/repeat_penalty/mean": 0.9524615177741418, "eval_rewards/repeat_penalty/std": 0.07457881277570358, "eval_rewards/total_composite/mean": 0.717106791642996, "eval_rewards/total_composite/std": 0.329087950862371, "eval_reward": 0.717106791642996, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.03292630620014209, "eval_sampling/sampling_logp_difference/max": 1.1617195422832782, "eval_sampling/importance_sampling_ratio/min": 0.32078430171196276, "eval_sampling/importance_sampling_ratio/mean": 1.0094741124373217, "eval_sampling/importance_sampling_ratio/max": 1.5328054244701679, "eval_entropy": 0.38292053456489855, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.717106791642996, "eval_reward_meter_mean": 0.7971470860334543, "eval_reward_meter_std": 0.31754180788993835, "eval_reward_count_adherence_mean": 0.9559012926541842, "eval_reward_count_adherence_std": 0.06501825956197885, "eval_reward_arabic_clean_mean": 0.9807692307692307, "eval_reward_arabic_clean_std": 0.05439282839114849, "eval_reward_repeat_penalty_mean": 0.9524615177741418, "eval_reward_repeat_penalty_std": 0.07457881277570358, "eval_reward_total_composite_mean": 0.717106791642996, "eval_reward_total_composite_std": 0.329087950862371} {"timestamp_utc": "2026-04-12T02:46:43Z", "mode": "train", "global_step": 2801, "epoch": 0.11250351447965619, "loss": -0.0039, "grad_norm": 2.5363826751708984, "learning_rate": 1.5151515151515152e-06, "num_tokens": 6362056.0, "completions/mean_length": 123.0, "completions/min_length": 121.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 123.0, "completions/min_terminated_length": 121.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.9976023435592651, "rewards/meter/std": 0.0004385724605526775, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.9619686603546143, "rewards/total_composite/std": 0.06589511781930923, "reward": 0.9619686603546143, "reward_std": 0.06589511036872864, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022641444578766823, "sampling/sampling_logp_difference/max": 1.1188087463378906, "sampling/importance_sampling_ratio/min": 0.3919609487056732, "sampling/importance_sampling_ratio/mean": 1.0046809911727905, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17738944850862026, "clip_ratio/low_mean": 0.004123763646930456, "clip_ratio/low_min": 0.004123763646930456, "clip_ratio/high_mean": 0.016170406714081764, "clip_ratio/high_max": 0.016170406714081764, "clip_ratio/region_mean": 0.02029417036101222, "reward_total_mean": 0.9619686603546143, "reward_meter_mean": 0.9976023435592651, "reward_meter_std": 0.0004385724605526775, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.9619686603546143, "reward_total_composite_std": 0.06589511781930923} {"timestamp_utc": "2026-04-12T02:46:49Z", "mode": "train", "global_step": 2802, "epoch": 0.11254367996144114, "loss": -0.0145, "grad_norm": 3.6220641136169434, "learning_rate": 1.5121212121212123e-06, "num_tokens": 6364387.0, "completions/mean_length": 135.375, "completions/min_length": 129.0, "completions/max_length": 147.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 135.375, "completions/min_terminated_length": 129.0, "completions/max_terminated_length": 147.0, "rewards/meter/mean": 0.8960581421852112, "rewards/meter/std": 0.2624220848083496, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8960581421852112, "rewards/total_composite/std": 0.2624220848083496, "reward": 0.8960581421852112, "reward_std": 0.2624220848083496, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.049931738525629044, "sampling/sampling_logp_difference/max": 1.0952783823013306, "sampling/importance_sampling_ratio/min": 0.35183805227279663, "sampling/importance_sampling_ratio/mean": 1.0152724981307983, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4941607192158699, "clip_ratio/low_mean": 0.0019379844889044762, "clip_ratio/low_min": 0.0019379844889044762, "clip_ratio/high_mean": 0.03208270261529833, "clip_ratio/high_max": 0.03208270261529833, "clip_ratio/region_mean": 0.03402068710420281, "reward_total_mean": 0.8960581421852112, "reward_meter_mean": 0.8960581421852112, "reward_meter_std": 0.2624220848083496, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8960581421852112, "reward_total_composite_std": 0.2624220848083496} {"timestamp_utc": "2026-04-12T02:46:53Z", "mode": "train", "global_step": 2803, "epoch": 0.1125838454432261, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.5090909090909091e-06, "num_tokens": 6366291.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00012427744513843209, "sampling/sampling_logp_difference/max": 0.00598335824906826, "sampling/importance_sampling_ratio/min": 0.9940345287322998, "sampling/importance_sampling_ratio/mean": 1.0000933408737183, "sampling/importance_sampling_ratio/max": 1.0050559043884277, "entropy": 0.0011047264706576243, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:47:00Z", "mode": "train", "global_step": 2804, "epoch": 0.11262401092501105, "loss": 0.0106, "grad_norm": 2.9486567974090576, "learning_rate": 1.5060606060606062e-06, "num_tokens": 6369524.0, "completions/mean_length": 235.125, "completions/min_length": 228.0, "completions/max_length": 241.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 235.125, "completions/min_terminated_length": 228.0, "completions/max_terminated_length": 241.0, "rewards/meter/mean": 0.9466480016708374, "rewards/meter/std": 0.14826621115207672, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9886363744735718, "rewards/repeat_penalty/std": 0.03214120864868164, "rewards/total_composite/mean": 0.9352947473526001, "rewards/total_composite/std": 0.14715024828910828, "reward": 0.9352947473526001, "reward_std": 0.14715023338794708, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04945862293243408, "sampling/sampling_logp_difference/max": 1.3091225624084473, "sampling/importance_sampling_ratio/min": 0.27005690336227417, "sampling/importance_sampling_ratio/mean": 1.0136529207229614, "sampling/importance_sampling_ratio/max": 1.872164011001587, "entropy": 0.46530257537961006, "clip_ratio/low_mean": 0.010395145742222667, "clip_ratio/low_min": 0.010395145742222667, "clip_ratio/high_mean": 0.026202646549791098, "clip_ratio/high_max": 0.026202646549791098, "clip_ratio/region_mean": 0.036597792292013764, "reward_total_mean": 0.9352947473526001, "reward_meter_mean": 0.9466480016708374, "reward_meter_std": 0.14826621115207672, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9886363744735718, "reward_repeat_penalty_std": 0.03214120864868164, "reward_total_composite_mean": 0.9352947473526001, "reward_total_composite_std": 0.14715024828910828} {"timestamp_utc": "2026-04-12T02:47:05Z", "mode": "train", "global_step": 2805, "epoch": 0.112664176406796, "loss": -0.0011, "grad_norm": 4.279446125030518, "learning_rate": 1.5030303030303032e-06, "num_tokens": 6371724.0, "completions/mean_length": 100.0, "completions/min_length": 97.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.0, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.9975219964981079, "rewards/meter/std": 0.0041835736483335495, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9728400707244873, "rewards/total_composite/std": 0.07395301014184952, "reward": 0.9728400707244873, "reward_std": 0.07395300269126892, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03659245744347572, "sampling/sampling_logp_difference/max": 1.4790048599243164, "sampling/importance_sampling_ratio/min": 0.2278643399477005, "sampling/importance_sampling_ratio/mean": 1.008768081665039, "sampling/importance_sampling_ratio/max": 1.6153602600097656, "entropy": 0.3125297725200653, "clip_ratio/low_mean": 0.0063775512389838696, "clip_ratio/low_min": 0.0063775512389838696, "clip_ratio/high_mean": 0.029976313933730125, "clip_ratio/high_max": 0.029976313933730125, "clip_ratio/region_mean": 0.036353865172713995, "reward_total_mean": 0.9728400707244873, "reward_meter_mean": 0.9975219964981079, "reward_meter_std": 0.0041835736483335495, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9728400707244873, "reward_total_composite_std": 0.07395301014184952} {"timestamp_utc": "2026-04-12T02:47:09Z", "mode": "train", "global_step": 2806, "epoch": 0.11270434188858096, "loss": -0.0041, "grad_norm": 0.8794472813606262, "learning_rate": 1.5e-06, "num_tokens": 6373380.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.7413170337677002, "rewards/meter/std": 0.13101767003536224, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7413170337677002, "rewards/total_composite/std": 0.13101767003536224, "reward": 0.7413170337677002, "reward_std": 0.13101768493652344, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0035149240866303444, "sampling/sampling_logp_difference/max": 1.4165477752685547, "sampling/importance_sampling_ratio/min": 0.24254991114139557, "sampling/importance_sampling_ratio/mean": 0.9984825849533081, "sampling/importance_sampling_ratio/max": 1.0023080110549927, "entropy": 0.0018152309057768434, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7413170337677002, "reward_meter_mean": 0.7413170337677002, "reward_meter_std": 0.13101767003536224, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7413170337677002, "reward_total_composite_std": 0.13101767003536224} {"timestamp_utc": "2026-04-12T02:47:14Z", "mode": "train", "global_step": 2807, "epoch": 0.11274450737036591, "loss": -0.0079, "grad_norm": 1.6524964570999146, "learning_rate": 1.496969696969697e-06, "num_tokens": 6375595.0, "completions/mean_length": 118.875, "completions/min_length": 116.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.875, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.9989545345306396, "rewards/meter/std": 0.00029157428070902824, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989545345306396, "rewards/total_composite/std": 0.00029157428070902824, "reward": 0.9989545345306396, "reward_std": 0.00029156444361433387, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03430765122175217, "sampling/sampling_logp_difference/max": 1.1520414352416992, "sampling/importance_sampling_ratio/min": 0.31599104404449463, "sampling/importance_sampling_ratio/mean": 1.0066179037094116, "sampling/importance_sampling_ratio/max": 1.6717981100082397, "entropy": 0.30837439745664597, "clip_ratio/low_mean": 0.004273504368029535, "clip_ratio/low_min": 0.004273504368029535, "clip_ratio/high_mean": 0.01668829272966832, "clip_ratio/high_max": 0.01668829272966832, "clip_ratio/region_mean": 0.020961797097697854, "reward_total_mean": 0.9989545345306396, "reward_meter_mean": 0.9989545345306396, "reward_meter_std": 0.00029157428070902824, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9989545345306396, "reward_total_composite_std": 0.00029157428070902824} {"timestamp_utc": "2026-04-12T02:47:19Z", "mode": "train", "global_step": 2808, "epoch": 0.11278467285215087, "loss": -0.0002, "grad_norm": 0.13850678503513336, "learning_rate": 1.493939393939394e-06, "num_tokens": 6377373.0, "completions/mean_length": 66.25, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.25, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981439709663391, "rewards/meter/std": 1.6234846043516882e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981439709663391, "rewards/total_composite/std": 1.6234846043516882e-05, "reward": 0.9981439709663391, "reward_std": 1.6221796613535844e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006040393374860287, "sampling/sampling_logp_difference/max": 0.7629547119140625, "sampling/importance_sampling_ratio/min": 0.46628665924072266, "sampling/importance_sampling_ratio/mean": 0.9976470470428467, "sampling/importance_sampling_ratio/max": 1.0641453266143799, "entropy": 0.022154160542413592, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/high_mean": 0.00562528264708817, "clip_ratio/high_max": 0.00562528264708817, "clip_ratio/region_mean": 0.007519222097471356, "reward_total_mean": 0.9981439709663391, "reward_meter_mean": 0.9981439709663391, "reward_meter_std": 1.6234846043516882e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981439709663391, "reward_total_composite_std": 1.6234846043516882e-05} {"timestamp_utc": "2026-04-12T02:47:24Z", "mode": "train", "global_step": 2809, "epoch": 0.11282483833393582, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.490909090909091e-06, "num_tokens": 6379350.0, "completions/mean_length": 66.125, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.125, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981522560119629, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981522560119629, "rewards/total_composite/std": 0.0, "reward": 0.9981522560119629, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0026704377960413694, "sampling/sampling_logp_difference/max": 0.33742189407348633, "sampling/importance_sampling_ratio/min": 0.7136077284812927, "sampling/importance_sampling_ratio/mean": 0.9995959401130676, "sampling/importance_sampling_ratio/max": 1.0592564344406128, "entropy": 0.015775780542753637, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9981522560119629, "reward_meter_mean": 0.9981522560119629, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981522560119629, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:47:28Z", "mode": "train", "global_step": 2810, "epoch": 0.11286500381572077, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.4878787878787878e-06, "num_tokens": 6380838.0, "completions/mean_length": 28.0, "completions/min_length": 28.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.0, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9943599104881287, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9943599104881287, "rewards/total_composite/std": 0.0, "reward": 0.9943599104881287, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00016406136273872107, "sampling/sampling_logp_difference/max": 0.0028465576469898224, "sampling/importance_sampling_ratio/min": 0.9998530745506287, "sampling/importance_sampling_ratio/mean": 1.0001628398895264, "sampling/importance_sampling_ratio/max": 1.0028506517410278, "entropy": 0.0013478934415616095, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9943599104881287, "reward_meter_mean": 0.9943599104881287, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9943599104881287, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:47:33Z", "mode": "train", "global_step": 2811, "epoch": 0.11290516929750573, "loss": 0.0043, "grad_norm": 4.882931232452393, "learning_rate": 1.484848484848485e-06, "num_tokens": 6382557.0, "completions/mean_length": 65.875, "completions/min_length": 65.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.875, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9871849417686462, "rewards/meter/std": 0.011940443888306618, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9871849417686462, "rewards/total_composite/std": 0.011940443888306618, "reward": 0.9871849417686462, "reward_std": 0.011940454132854939, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.029961178079247475, "sampling/sampling_logp_difference/max": 1.0541696548461914, "sampling/importance_sampling_ratio/min": 0.3484816551208496, "sampling/importance_sampling_ratio/mean": 1.0042191743850708, "sampling/importance_sampling_ratio/max": 1.5272610187530518, "entropy": 0.25444527715444565, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/high_mean": 0.013288452988490462, "clip_ratio/high_max": 0.013288452988490462, "clip_ratio/region_mean": 0.015182392438873649, "reward_total_mean": 0.9871849417686462, "reward_meter_mean": 0.9871849417686462, "reward_meter_std": 0.011940443888306618, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9871849417686462, "reward_total_composite_std": 0.011940443888306618} {"timestamp_utc": "2026-04-12T02:47:37Z", "mode": "train", "global_step": 2812, "epoch": 0.11294533477929068, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.481818181818182e-06, "num_tokens": 6383901.0, "completions/mean_length": 28.0, "completions/min_length": 28.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.0, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9943599104881287, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9943599104881287, "rewards/total_composite/std": 0.0, "reward": 0.9943599104881287, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00016283367585856467, "sampling/sampling_logp_difference/max": 0.0031713712960481644, "sampling/importance_sampling_ratio/min": 0.9999763369560242, "sampling/importance_sampling_ratio/mean": 1.0001628398895264, "sampling/importance_sampling_ratio/max": 1.0031764507293701, "entropy": 0.0012169194815214723, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9943599104881287, "reward_meter_mean": 0.9943599104881287, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9943599104881287, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:47:46Z", "mode": "train", "global_step": 2813, "epoch": 0.11298550026107564, "loss": -0.0216, "grad_norm": 1.4151586294174194, "learning_rate": 1.478787878787879e-06, "num_tokens": 6388632.0, "completions/mean_length": 402.375, "completions/min_length": 390.0, "completions/max_length": 438.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 402.375, "completions/min_terminated_length": 390.0, "completions/max_terminated_length": 438.0, "rewards/meter/mean": 0.9989125728607178, "rewards/meter/std": 0.0006985744112171233, "rewards/count_adherence/mean": 0.7788461446762085, "rewards/count_adherence/std": 0.027196412906050682, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9802631139755249, "rewards/repeat_penalty/std": 0.027239417657256126, "rewards/total_composite/mean": 0.7628299593925476, "rewards/total_composite/std": 0.0387784019112587, "reward": 0.7628299593925476, "reward_std": 0.038778405636548996, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04369629919528961, "sampling/sampling_logp_difference/max": 2.0083513259887695, "sampling/importance_sampling_ratio/min": 0.1342097669839859, "sampling/importance_sampling_ratio/mean": 1.010063886642456, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43715671822428703, "clip_ratio/low_mean": 0.008746044244617224, "clip_ratio/low_min": 0.008746044244617224, "clip_ratio/high_mean": 0.02115830732509494, "clip_ratio/high_max": 0.02115830732509494, "clip_ratio/region_mean": 0.029904351569712162, "reward_total_mean": 0.7628299593925476, "reward_meter_mean": 0.9989125728607178, "reward_meter_std": 0.0006985744112171233, "reward_count_adherence_mean": 0.7788461446762085, "reward_count_adherence_std": 0.027196412906050682, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9802631139755249, "reward_repeat_penalty_std": 0.027239417657256126, "reward_total_composite_mean": 0.7628299593925476, "reward_total_composite_std": 0.0387784019112587} {"timestamp_utc": "2026-04-12T02:47:52Z", "mode": "train", "global_step": 2814, "epoch": 0.11302566574286059, "loss": -0.0018, "grad_norm": 1.2337918281555176, "learning_rate": 1.475757575757576e-06, "num_tokens": 6391131.0, "completions/mean_length": 127.375, "completions/min_length": 125.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.375, "completions/min_terminated_length": 125.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.9961709976196289, "rewards/meter/std": 0.0034113116562366486, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9428157210350037, "rewards/total_composite/std": 0.07388003170490265, "reward": 0.9428157210350037, "reward_std": 0.07388003170490265, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012026426382362843, "sampling/sampling_logp_difference/max": 1.767538070678711, "sampling/importance_sampling_ratio/min": 0.1707528531551361, "sampling/importance_sampling_ratio/mean": 1.0020824670791626, "sampling/importance_sampling_ratio/max": 1.6902966499328613, "entropy": 0.11378153134137392, "clip_ratio/low_mean": 0.0029606895986944437, "clip_ratio/low_min": 0.0029606895986944437, "clip_ratio/high_mean": 0.008882812689989805, "clip_ratio/high_max": 0.008882812689989805, "clip_ratio/region_mean": 0.011843502288684249, "reward_total_mean": 0.9428157210350037, "reward_meter_mean": 0.9961709976196289, "reward_meter_std": 0.0034113116562366486, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.9428157210350037, "reward_total_composite_std": 0.07388003170490265} {"timestamp_utc": "2026-04-12T02:47:56Z", "mode": "train", "global_step": 2815, "epoch": 0.11306583122464554, "loss": 0.013, "grad_norm": 1.3737421035766602, "learning_rate": 1.4727272727272728e-06, "num_tokens": 6392648.0, "completions/mean_length": 40.625, "completions/min_length": 39.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.625, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.99871426820755, "rewards/meter/std": 0.0005067276651971042, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.99871426820755, "rewards/total_composite/std": 0.0005067276651971042, "reward": 0.99871426820755, "reward_std": 0.0005067184683866799, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01035021897405386, "sampling/sampling_logp_difference/max": 0.8471136093139648, "sampling/importance_sampling_ratio/min": 0.4286504089832306, "sampling/importance_sampling_ratio/mean": 0.9997342824935913, "sampling/importance_sampling_ratio/max": 1.4657316207885742, "entropy": 0.0865395087748766, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.006330128293484449, "clip_ratio/high_max": 0.006330128293484449, "clip_ratio/region_mean": 0.006330128293484449, "reward_total_mean": 0.99871426820755, "reward_meter_mean": 0.99871426820755, "reward_meter_std": 0.0005067276651971042, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.99871426820755, "reward_total_composite_std": 0.0005067276651971042} {"timestamp_utc": "2026-04-12T02:48:01Z", "mode": "train", "global_step": 2816, "epoch": 0.1131059967064305, "loss": 0.0064, "grad_norm": 3.449428081512451, "learning_rate": 1.4696969696969698e-06, "num_tokens": 6394769.0, "completions/mean_length": 102.125, "completions/min_length": 101.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.125, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.9987610578536987, "rewards/meter/std": 0.0006801035488024354, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9987610578536987, "rewards/total_composite/std": 0.0006801035488024354, "reward": 0.9987610578536987, "reward_std": 0.0006801035488024354, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03967723622918129, "sampling/sampling_logp_difference/max": 1.2159053087234497, "sampling/importance_sampling_ratio/min": 0.29644152522087097, "sampling/importance_sampling_ratio/mean": 1.0010402202606201, "sampling/importance_sampling_ratio/max": 1.9102437496185303, "entropy": 0.33477963879704475, "clip_ratio/low_mean": 0.01341109094209969, "clip_ratio/low_min": 0.01341109094209969, "clip_ratio/high_mean": 0.022014267276972532, "clip_ratio/high_max": 0.022014267276972532, "clip_ratio/region_mean": 0.03542535821907222, "reward_total_mean": 0.9987610578536987, "reward_meter_mean": 0.9987610578536987, "reward_meter_std": 0.0006801035488024354, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9987610578536987, "reward_total_composite_std": 0.0006801035488024354} {"timestamp_utc": "2026-04-12T02:48:05Z", "mode": "train", "global_step": 2817, "epoch": 0.11314616218821545, "loss": 0.0051, "grad_norm": 4.832633972167969, "learning_rate": 1.4666666666666669e-06, "num_tokens": 6396562.0, "completions/mean_length": 67.125, "completions/min_length": 66.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.125, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9991535544395447, "rewards/meter/std": 0.00035637259134091437, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991535544395447, "rewards/total_composite/std": 0.00035637259134091437, "reward": 0.9991535544395447, "reward_std": 0.00035637334804050624, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03295252099633217, "sampling/sampling_logp_difference/max": 5.969149589538574, "sampling/importance_sampling_ratio/min": 0.0025564145762473345, "sampling/importance_sampling_ratio/mean": 0.9987229108810425, "sampling/importance_sampling_ratio/max": 1.4108344316482544, "entropy": 0.15694944839924574, "clip_ratio/low_mean": 0.0037313431967049837, "clip_ratio/low_min": 0.0037313431967049837, "clip_ratio/high_mean": 0.01860184350516647, "clip_ratio/high_max": 0.01860184350516647, "clip_ratio/region_mean": 0.022333186701871455, "reward_total_mean": 0.9991535544395447, "reward_meter_mean": 0.9991535544395447, "reward_meter_std": 0.00035637259134091437, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9991535544395447, "reward_total_composite_std": 0.00035637259134091437} {"timestamp_utc": "2026-04-12T02:48:12Z", "mode": "train", "global_step": 2818, "epoch": 0.1131863276700004, "loss": -0.002, "grad_norm": 1.7950763702392578, "learning_rate": 1.4636363636363637e-06, "num_tokens": 6399800.0, "completions/mean_length": 198.75, "completions/min_length": 196.0, "completions/max_length": 200.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 198.75, "completions/min_terminated_length": 196.0, "completions/max_terminated_length": 200.0, "rewards/meter/mean": 0.9993983507156372, "rewards/meter/std": 7.34872737666592e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9090909361839294, "rewards/repeat_penalty/std": 0.048592954874038696, "rewards/total_composite/mean": 0.9085431098937988, "rewards/total_composite/std": 0.04854762554168701, "reward": 0.9085431098937988, "reward_std": 0.04854761064052582, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014119832776486874, "sampling/sampling_logp_difference/max": 1.0774822235107422, "sampling/importance_sampling_ratio/min": 0.3404516279697418, "sampling/importance_sampling_ratio/mean": 1.003812551498413, "sampling/importance_sampling_ratio/max": 1.756600022315979, "entropy": 0.12045034673064947, "clip_ratio/low_mean": 0.0075952449114993215, "clip_ratio/low_min": 0.0075952449114993215, "clip_ratio/high_mean": 0.007499999832361937, "clip_ratio/high_max": 0.007499999832361937, "clip_ratio/region_mean": 0.015095244743861258, "reward_total_mean": 0.9085431098937988, "reward_meter_mean": 0.9993983507156372, "reward_meter_std": 7.34872737666592e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9090909361839294, "reward_repeat_penalty_std": 0.048592954874038696, "reward_total_composite_mean": 0.9085431098937988, "reward_total_composite_std": 0.04854762554168701} {"timestamp_utc": "2026-04-12T02:48:17Z", "mode": "train", "global_step": 2819, "epoch": 0.11322649315178536, "loss": -0.0006, "grad_norm": 4.404444217681885, "learning_rate": 1.4606060606060608e-06, "num_tokens": 6402006.0, "completions/mean_length": 106.75, "completions/min_length": 106.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.75, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.9992326498031616, "rewards/meter/std": 0.0003244501131121069, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992326498031616, "rewards/total_composite/std": 0.0003244501131121069, "reward": 0.9992326498031616, "reward_std": 0.00032444868702441454, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01700535975396633, "sampling/sampling_logp_difference/max": 1.6258955001831055, "sampling/importance_sampling_ratio/min": 0.1967354267835617, "sampling/importance_sampling_ratio/mean": 1.0044901371002197, "sampling/importance_sampling_ratio/max": 1.5127817392349243, "entropy": 0.12008912675082684, "clip_ratio/low_mean": 0.003537735901772976, "clip_ratio/low_min": 0.003537735901772976, "clip_ratio/high_mean": 0.007020366610959172, "clip_ratio/high_max": 0.007020366610959172, "clip_ratio/region_mean": 0.010558102512732148, "reward_total_mean": 0.9992326498031616, "reward_meter_mean": 0.9992326498031616, "reward_meter_std": 0.0003244501131121069, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992326498031616, "reward_total_composite_std": 0.0003244501131121069} {"timestamp_utc": "2026-04-12T02:48:21Z", "mode": "train", "global_step": 2820, "epoch": 0.11326665863357031, "loss": -0.0417, "grad_norm": 4.645050525665283, "learning_rate": 1.4575757575757576e-06, "num_tokens": 6403787.0, "completions/mean_length": 69.625, "completions/min_length": 62.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.625, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9988994598388672, "rewards/meter/std": 0.001365774660371244, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9988994598388672, "rewards/total_composite/std": 0.001365774660371244, "reward": 0.9988994598388672, "reward_std": 0.0013657818781211972, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022290274500846863, "sampling/sampling_logp_difference/max": 1.9662590026855469, "sampling/importance_sampling_ratio/min": 0.1399795413017273, "sampling/importance_sampling_ratio/mean": 0.9964150786399841, "sampling/importance_sampling_ratio/max": 1.4191588163375854, "entropy": 0.10764285689219832, "clip_ratio/low_mean": 0.008064515888690948, "clip_ratio/low_min": 0.008064515888690948, "clip_ratio/high_mean": 0.00890342053025961, "clip_ratio/high_max": 0.00890342053025961, "clip_ratio/region_mean": 0.016967936418950558, "reward_total_mean": 0.9988994598388672, "reward_meter_mean": 0.9988994598388672, "reward_meter_std": 0.001365774660371244, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9988994598388672, "reward_total_composite_std": 0.001365774660371244} {"timestamp_utc": "2026-04-12T02:48:25Z", "mode": "train", "global_step": 2821, "epoch": 0.11330682411535527, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.4545454545454546e-06, "num_tokens": 6405779.0, "completions/mean_length": 80.0, "completions/min_length": 80.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.0, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "reward": 0.7190229296684265, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00021103888866491616, "sampling/sampling_logp_difference/max": 0.002838927786797285, "sampling/importance_sampling_ratio/min": 0.9997273683547974, "sampling/importance_sampling_ratio/mean": 1.0002089738845825, "sampling/importance_sampling_ratio/max": 1.0028430223464966, "entropy": 0.0020125040755374357, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7190229296684265, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:48:31Z", "mode": "train", "global_step": 2822, "epoch": 0.11334698959714022, "loss": -0.002, "grad_norm": 1.043582558631897, "learning_rate": 1.4515151515151515e-06, "num_tokens": 6408280.0, "completions/mean_length": 158.625, "completions/min_length": 157.0, "completions/max_length": 160.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 158.625, "completions/min_terminated_length": 157.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.980668306350708, "rewards/meter/std": 0.04880243167281151, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9444444179534912, "rewards/repeat_penalty/std": 0.059391383081674576, "rewards/total_composite/mean": 0.9252284169197083, "rewards/total_composite/std": 0.0608980655670166, "reward": 0.9252284169197083, "reward_std": 0.0608980655670166, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013364696875214577, "sampling/sampling_logp_difference/max": 1.1392550468444824, "sampling/importance_sampling_ratio/min": 0.32005739212036133, "sampling/importance_sampling_ratio/mean": 1.0040552616119385, "sampling/importance_sampling_ratio/max": 1.4671419858932495, "entropy": 0.14032841473817825, "clip_ratio/low_mean": 0.003149755357299, "clip_ratio/low_min": 0.003149755357299, "clip_ratio/high_mean": 0.003930817823857069, "clip_ratio/high_max": 0.003930817823857069, "clip_ratio/region_mean": 0.007080573181156069, "reward_total_mean": 0.9252284169197083, "reward_meter_mean": 0.980668306350708, "reward_meter_std": 0.04880243167281151, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9444444179534912, "reward_repeat_penalty_std": 0.059391383081674576, "reward_total_composite_mean": 0.9252284169197083, "reward_total_composite_std": 0.0608980655670166} {"timestamp_utc": "2026-04-12T02:48:35Z", "mode": "train", "global_step": 2823, "epoch": 0.11338715507892518, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.4484848484848485e-06, "num_tokens": 6409984.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002062379935523495, "sampling/sampling_logp_difference/max": 0.0020993247162550688, "sampling/importance_sampling_ratio/min": 0.9997925162315369, "sampling/importance_sampling_ratio/mean": 1.000205397605896, "sampling/importance_sampling_ratio/max": 1.0021015405654907, "entropy": 0.001565092708915472, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:48:40Z", "mode": "train", "global_step": 2824, "epoch": 0.11342732056071013, "loss": 0.0014, "grad_norm": 0.7708361148834229, "learning_rate": 1.4454545454545453e-06, "num_tokens": 6411779.0, "completions/mean_length": 71.375, "completions/min_length": 71.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.375, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9994142055511475, "rewards/meter/std": 5.688391684088856e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994142055511475, "rewards/total_composite/std": 5.688391684088856e-05, "reward": 0.9994142055511475, "reward_std": 5.688605597242713e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009274118579924107, "sampling/sampling_logp_difference/max": 0.677030086517334, "sampling/importance_sampling_ratio/min": 0.5081238150596619, "sampling/importance_sampling_ratio/mean": 1.0028364658355713, "sampling/importance_sampling_ratio/max": 1.4751698970794678, "entropy": 0.06790398666635156, "clip_ratio/low_mean": 0.0017361111240461469, "clip_ratio/low_min": 0.0017361111240461469, "clip_ratio/high_mean": 0.0017361111240461469, "clip_ratio/high_max": 0.0017361111240461469, "clip_ratio/region_mean": 0.0034722222480922937, "reward_total_mean": 0.9994142055511475, "reward_meter_mean": 0.9994142055511475, "reward_meter_std": 5.688391684088856e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994142055511475, "reward_total_composite_std": 5.688391684088856e-05} {"timestamp_utc": "2026-04-12T02:48:45Z", "mode": "train", "global_step": 2825, "epoch": 0.11346748604249508, "loss": -0.0007, "grad_norm": 0.8596204519271851, "learning_rate": 1.4424242424242426e-06, "num_tokens": 6413966.0, "completions/mean_length": 106.375, "completions/min_length": 104.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.375, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.9992821216583252, "rewards/meter/std": 0.0001443348592147231, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992821216583252, "rewards/total_composite/std": 0.0001443348592147231, "reward": 0.9992821216583252, "reward_std": 0.000144336445373483, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018452905118465424, "sampling/sampling_logp_difference/max": 1.3011884689331055, "sampling/importance_sampling_ratio/min": 0.2722080945968628, "sampling/importance_sampling_ratio/mean": 1.001145839691162, "sampling/importance_sampling_ratio/max": 1.6910252571105957, "entropy": 0.12096875347197056, "clip_ratio/low_mean": 0.0035493926843628287, "clip_ratio/low_min": 0.0035493926843628287, "clip_ratio/high_mean": 0.009345794096589088, "clip_ratio/high_max": 0.009345794096589088, "clip_ratio/region_mean": 0.012895186780951917, "reward_total_mean": 0.9992821216583252, "reward_meter_mean": 0.9992821216583252, "reward_meter_std": 0.0001443348592147231, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992821216583252, "reward_total_composite_std": 0.0001443348592147231} {"timestamp_utc": "2026-04-12T02:48:50Z", "mode": "train", "global_step": 2826, "epoch": 0.11350765152428004, "loss": -0.0004, "grad_norm": 0.9645191431045532, "learning_rate": 1.4393939393939396e-06, "num_tokens": 6415713.0, "completions/mean_length": 66.375, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.375, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981287717819214, "rewards/meter/std": 3.6517179978545755e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981287717819214, "rewards/total_composite/std": 3.6517179978545755e-05, "reward": 0.9981287717819214, "reward_std": 3.651110455393791e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007696997374296188, "sampling/sampling_logp_difference/max": 0.9785857200622559, "sampling/importance_sampling_ratio/min": 0.37584227323532104, "sampling/importance_sampling_ratio/mean": 0.9988413453102112, "sampling/importance_sampling_ratio/max": 1.138381838798523, "entropy": 0.03601772431284189, "clip_ratio/low_mean": 0.0037313431967049837, "clip_ratio/low_min": 0.0037313431967049837, "clip_ratio/high_mean": 0.0018656715983524919, "clip_ratio/high_max": 0.0018656715983524919, "clip_ratio/region_mean": 0.005597014795057476, "reward_total_mean": 0.9981287717819214, "reward_meter_mean": 0.9981287717819214, "reward_meter_std": 3.6517179978545755e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981287717819214, "reward_total_composite_std": 3.6517179978545755e-05} {"timestamp_utc": "2026-04-12T02:48:56Z", "mode": "train", "global_step": 2827, "epoch": 0.11354781700606499, "loss": -0.0018, "grad_norm": 2.085141897201538, "learning_rate": 1.4363636363636365e-06, "num_tokens": 6418795.0, "completions/mean_length": 197.25, "completions/min_length": 192.0, "completions/max_length": 201.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 197.25, "completions/min_terminated_length": 192.0, "completions/max_terminated_length": 201.0, "rewards/meter/mean": 0.9994193315505981, "rewards/meter/std": 5.9720292483689263e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9545454978942871, "rewards/repeat_penalty/std": 0.0485929399728775, "rewards/total_composite/mean": 0.9539901614189148, "rewards/total_composite/std": 0.04854191467165947, "reward": 0.9539901614189148, "reward_std": 0.04854191467165947, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015652773901820183, "sampling/sampling_logp_difference/max": 1.3481910228729248, "sampling/importance_sampling_ratio/min": 0.2597096562385559, "sampling/importance_sampling_ratio/mean": 1.0017954111099243, "sampling/importance_sampling_ratio/max": 1.5125662088394165, "entropy": 0.117509332485497, "clip_ratio/low_mean": 0.005142431997228414, "clip_ratio/low_min": 0.005142431997228414, "clip_ratio/high_mean": 0.008849852601997554, "clip_ratio/high_max": 0.008849852601997554, "clip_ratio/region_mean": 0.013992284599225968, "reward_total_mean": 0.9539901614189148, "reward_meter_mean": 0.9994193315505981, "reward_meter_std": 5.9720292483689263e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9545454978942871, "reward_repeat_penalty_std": 0.0485929399728775, "reward_total_composite_mean": 0.9539901614189148, "reward_total_composite_std": 0.04854191467165947} {"timestamp_utc": "2026-04-12T02:49:00Z", "mode": "train", "global_step": 2828, "epoch": 0.11358798248784995, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.4333333333333335e-06, "num_tokens": 6420507.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00012123679334763438, "sampling/sampling_logp_difference/max": 0.008353465236723423, "sampling/importance_sampling_ratio/min": 0.9991452097892761, "sampling/importance_sampling_ratio/mean": 1.0001118183135986, "sampling/importance_sampling_ratio/max": 1.0083885192871094, "entropy": 0.0012116766738472506, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:49:05Z", "mode": "train", "global_step": 2829, "epoch": 0.1136281479696349, "loss": 0.0003, "grad_norm": 1.4268546104431152, "learning_rate": 1.4303030303030306e-06, "num_tokens": 6422684.0, "completions/mean_length": 93.125, "completions/min_length": 93.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.125, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.9977059364318848, "rewards/meter/std": 6.68626744300127e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977059364318848, "rewards/total_composite/std": 6.68626744300127e-05, "reward": 0.9977059364318848, "reward_std": 6.687968561891466e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0164635069668293, "sampling/sampling_logp_difference/max": 1.0106797218322754, "sampling/importance_sampling_ratio/min": 0.3639715313911438, "sampling/importance_sampling_ratio/mean": 1.0083869695663452, "sampling/importance_sampling_ratio/max": 1.625991702079773, "entropy": 0.10609597992151976, "clip_ratio/low_mean": 0.014784946106374264, "clip_ratio/low_min": 0.014784946106374264, "clip_ratio/high_mean": 0.0013297871919348836, "clip_ratio/high_max": 0.0013297871919348836, "clip_ratio/region_mean": 0.016114733298309147, "reward_total_mean": 0.9977059364318848, "reward_meter_mean": 0.9977059364318848, "reward_meter_std": 6.68626744300127e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9977059364318848, "reward_total_composite_std": 6.68626744300127e-05} {"timestamp_utc": "2026-04-12T02:49:11Z", "mode": "train", "global_step": 2830, "epoch": 0.11366831345141985, "loss": -0.0018, "grad_norm": 2.387721538543701, "learning_rate": 1.4272727272727274e-06, "num_tokens": 6425415.0, "completions/mean_length": 158.375, "completions/min_length": 157.0, "completions/max_length": 160.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 158.375, "completions/min_terminated_length": 157.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.9977245330810547, "rewards/meter/std": 0.0005517599638551474, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333134651184, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.9561665058135986, "rewards/total_composite/std": 0.05763240531086922, "reward": 0.9561665058135986, "reward_std": 0.05763239413499832, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01837637834250927, "sampling/sampling_logp_difference/max": 1.528066635131836, "sampling/importance_sampling_ratio/min": 0.21695472300052643, "sampling/importance_sampling_ratio/mean": 1.0021953582763672, "sampling/importance_sampling_ratio/max": 1.5251771211624146, "entropy": 0.16988196410238743, "clip_ratio/low_mean": 0.004751874483190477, "clip_ratio/low_min": 0.004751874483190477, "clip_ratio/high_mean": 0.010205571772530675, "clip_ratio/high_max": 0.010205571772530675, "clip_ratio/region_mean": 0.014957446255721152, "reward_total_mean": 0.9561665058135986, "reward_meter_mean": 0.9977245330810547, "reward_meter_std": 0.0005517599638551474, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333134651184, "reward_repeat_penalty_std": 0.05750546231865883, "reward_total_composite_mean": 0.9561665058135986, "reward_total_composite_std": 0.05763240531086922} {"timestamp_utc": "2026-04-12T02:49:16Z", "mode": "train", "global_step": 2831, "epoch": 0.1137084789332048, "loss": -0.0056, "grad_norm": 3.8205645084381104, "learning_rate": 1.4242424242424244e-06, "num_tokens": 6427666.0, "completions/mean_length": 106.375, "completions/min_length": 104.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.375, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.8774728775024414, "rewards/meter/std": 0.3444782793521881, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8774728775024414, "rewards/total_composite/std": 0.3444782793521881, "reward": 0.8774728775024414, "reward_std": 0.3444782495498657, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01733158528804779, "sampling/sampling_logp_difference/max": 0.9781675338745117, "sampling/importance_sampling_ratio/min": 0.37599948048591614, "sampling/importance_sampling_ratio/mean": 1.0037299394607544, "sampling/importance_sampling_ratio/max": 1.4661322832107544, "entropy": 0.1615914311259985, "clip_ratio/low_mean": 0.0072115384973585606, "clip_ratio/low_min": 0.0072115384973585606, "clip_ratio/high_mean": 0.008177978219464421, "clip_ratio/high_max": 0.008177978219464421, "clip_ratio/region_mean": 0.015389516716822982, "reward_total_mean": 0.8774728775024414, "reward_meter_mean": 0.8774728775024414, "reward_meter_std": 0.3444782793521881, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8774728775024414, "reward_total_composite_std": 0.3444782793521881} {"timestamp_utc": "2026-04-12T02:49:21Z", "mode": "train", "global_step": 2832, "epoch": 0.11374864441498976, "loss": 0.0047, "grad_norm": 0.9516367316246033, "learning_rate": 1.4212121212121213e-06, "num_tokens": 6429513.0, "completions/mean_length": 79.875, "completions/min_length": 79.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.875, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.7240880727767944, "rewards/meter/std": 0.014326409436762333, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7240880727767944, "rewards/total_composite/std": 0.014326409436762333, "reward": 0.7240880727767944, "reward_std": 0.014326424337923527, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.002547384472563863, "sampling/sampling_logp_difference/max": 1.0646743774414062, "sampling/importance_sampling_ratio/min": 0.3448401391506195, "sampling/importance_sampling_ratio/mean": 0.9994467496871948, "sampling/importance_sampling_ratio/max": 1.1789082288742065, "entropy": 0.007047486782539636, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0015822785208001733, "clip_ratio/high_max": 0.0015822785208001733, "clip_ratio/region_mean": 0.0015822785208001733, "reward_total_mean": 0.7240880727767944, "reward_meter_mean": 0.7240880727767944, "reward_meter_std": 0.014326409436762333, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7240880727767944, "reward_total_composite_std": 0.014326409436762333} {"timestamp_utc": "2026-04-12T02:49:25Z", "mode": "train", "global_step": 2833, "epoch": 0.11378880989677471, "loss": 0.0162, "grad_norm": 4.6605939865112305, "learning_rate": 1.4181818181818183e-06, "num_tokens": 6430923.0, "completions/mean_length": 35.25, "completions/min_length": 34.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.25, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.9975078105926514, "rewards/meter/std": 0.0002991736400872469, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9975078105926514, "rewards/total_composite/std": 0.0002991736400872469, "reward": 0.9975078105926514, "reward_std": 0.00029916909988969564, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023440107703208923, "sampling/sampling_logp_difference/max": 1.062718391418457, "sampling/importance_sampling_ratio/min": 0.34551531076431274, "sampling/importance_sampling_ratio/mean": 0.9931513667106628, "sampling/importance_sampling_ratio/max": 1.2360522747039795, "entropy": 0.12462522741407156, "clip_ratio/low_mean": 0.003289473708719015, "clip_ratio/low_min": 0.003289473708719015, "clip_ratio/high_mean": 0.010135134682059288, "clip_ratio/high_max": 0.010135134682059288, "clip_ratio/region_mean": 0.013424608390778303, "reward_total_mean": 0.9975078105926514, "reward_meter_mean": 0.9975078105926514, "reward_meter_std": 0.0002991736400872469, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9975078105926514, "reward_total_composite_std": 0.0002991736400872469} {"timestamp_utc": "2026-04-12T02:49:31Z", "mode": "train", "global_step": 2834, "epoch": 0.11382897537855967, "loss": 0.0036, "grad_norm": 1.4452556371688843, "learning_rate": 1.4151515151515151e-06, "num_tokens": 6433732.0, "completions/mean_length": 158.125, "completions/min_length": 155.0, "completions/max_length": 161.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 158.125, "completions/min_terminated_length": 155.0, "completions/max_terminated_length": 161.0, "rewards/meter/mean": 0.9938750267028809, "rewards/meter/std": 0.01016020867973566, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333134651184, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.9522948265075684, "rewards/total_composite/std": 0.05487677827477455, "reward": 0.9522948265075684, "reward_std": 0.054876767098903656, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021103786304593086, "sampling/sampling_logp_difference/max": 1.3641910552978516, "sampling/importance_sampling_ratio/min": 0.2555873394012451, "sampling/importance_sampling_ratio/mean": 1.001399040222168, "sampling/importance_sampling_ratio/max": 1.7436052560806274, "entropy": 0.19695778377354145, "clip_ratio/low_mean": 0.0023785202065482736, "clip_ratio/low_min": 0.0023785202065482736, "clip_ratio/high_mean": 0.010262453462928534, "clip_ratio/high_max": 0.010262453462928534, "clip_ratio/region_mean": 0.012640973669476807, "reward_total_mean": 0.9522948265075684, "reward_meter_mean": 0.9938750267028809, "reward_meter_std": 0.01016020867973566, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333134651184, "reward_repeat_penalty_std": 0.05750546231865883, "reward_total_composite_mean": 0.9522948265075684, "reward_total_composite_std": 0.05487677827477455} {"timestamp_utc": "2026-04-12T02:49:35Z", "mode": "train", "global_step": 2835, "epoch": 0.11386914086034462, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.4121212121212122e-06, "num_tokens": 6435500.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00019080200581811368, "sampling/sampling_logp_difference/max": 0.009624381549656391, "sampling/importance_sampling_ratio/min": 0.9914776682853699, "sampling/importance_sampling_ratio/mean": 1.0001351833343506, "sampling/importance_sampling_ratio/max": 1.0096708536148071, "entropy": 0.0018191997078247368, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:49:40Z", "mode": "train", "global_step": 2836, "epoch": 0.11390930634212958, "loss": -0.0092, "grad_norm": 3.5270473957061768, "learning_rate": 1.409090909090909e-06, "num_tokens": 6437557.0, "completions/mean_length": 97.125, "completions/min_length": 94.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.125, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.9715797901153564, "rewards/meter/std": 0.03168648108839989, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9715797901153564, "rewards/total_composite/std": 0.03168648108839989, "reward": 0.9715797901153564, "reward_std": 0.03168648108839989, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.047103382647037506, "sampling/sampling_logp_difference/max": 1.1195144653320312, "sampling/importance_sampling_ratio/min": 0.3264382481575012, "sampling/importance_sampling_ratio/mean": 1.0144469738006592, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4101872891187668, "clip_ratio/low_mean": 0.006538120680488646, "clip_ratio/low_min": 0.006538120680488646, "clip_ratio/high_mean": 0.014033293817192316, "clip_ratio/high_max": 0.014033293817192316, "clip_ratio/region_mean": 0.020571414497680962, "reward_total_mean": 0.9715797901153564, "reward_meter_mean": 0.9715797901153564, "reward_meter_std": 0.03168648108839989, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9715797901153564, "reward_total_composite_std": 0.03168648108839989} {"timestamp_utc": "2026-04-12T02:49:46Z", "mode": "train", "global_step": 2837, "epoch": 0.11394947182391453, "loss": 0.0116, "grad_norm": 2.079005002975464, "learning_rate": 1.406060606060606e-06, "num_tokens": 6440330.0, "completions/mean_length": 155.625, "completions/min_length": 151.0, "completions/max_length": 158.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 155.625, "completions/min_terminated_length": 151.0, "completions/max_terminated_length": 158.0, "rewards/meter/mean": 0.9973591566085815, "rewards/meter/std": 0.001539723016321659, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.9696353077888489, "rewards/total_composite/std": 0.05092417448759079, "reward": 0.9696353077888489, "reward_std": 0.05092417448759079, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0309575367718935, "sampling/sampling_logp_difference/max": 2.1312127113342285, "sampling/importance_sampling_ratio/min": 0.1186932623386383, "sampling/importance_sampling_ratio/mean": 1.0057966709136963, "sampling/importance_sampling_ratio/max": 1.7507938146591187, "entropy": 0.23529152572155, "clip_ratio/low_mean": 0.0031847134232521057, "clip_ratio/low_min": 0.0031847134232521057, "clip_ratio/high_mean": 0.02511692512780428, "clip_ratio/high_max": 0.02511692512780428, "clip_ratio/region_mean": 0.028301638551056385, "reward_total_mean": 0.9696353077888489, "reward_meter_mean": 0.9973591566085815, "reward_meter_std": 0.001539723016321659, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_total_composite_mean": 0.9696353077888489, "reward_total_composite_std": 0.05092417448759079} {"timestamp_utc": "2026-04-12T02:49:51Z", "mode": "train", "global_step": 2838, "epoch": 0.11398963730569948, "loss": -0.0031, "grad_norm": 2.711418628692627, "learning_rate": 1.403030303030303e-06, "num_tokens": 6442782.0, "completions/mean_length": 134.5, "completions/min_length": 132.0, "completions/max_length": 137.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 134.5, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.999076247215271, "rewards/meter/std": 0.00021775417553726584, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999076247215271, "rewards/total_composite/std": 0.00021775417553726584, "reward": 0.999076247215271, "reward_std": 0.00021775579079985619, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.044978611171245575, "sampling/sampling_logp_difference/max": 1.342717170715332, "sampling/importance_sampling_ratio/min": 0.26113519072532654, "sampling/importance_sampling_ratio/mean": 1.0005239248275757, "sampling/importance_sampling_ratio/max": 1.565982460975647, "entropy": 0.37597112730145454, "clip_ratio/low_mean": 0.023296055383980274, "clip_ratio/low_min": 0.023296055383980274, "clip_ratio/high_mean": 0.012895984342321754, "clip_ratio/high_max": 0.012895984342321754, "clip_ratio/region_mean": 0.03619203972630203, "reward_total_mean": 0.999076247215271, "reward_meter_mean": 0.999076247215271, "reward_meter_std": 0.00021775417553726584, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.999076247215271, "reward_total_composite_std": 0.00021775417553726584} {"timestamp_utc": "2026-04-12T02:49:55Z", "mode": "train", "global_step": 2839, "epoch": 0.11402980278748444, "loss": 0.0133, "grad_norm": 4.530300140380859, "learning_rate": 1.4000000000000001e-06, "num_tokens": 6444599.0, "completions/mean_length": 66.125, "completions/min_length": 65.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.125, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9489889144897461, "rewards/meter/std": 0.003239382989704609, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9253357648849487, "rewards/total_composite/std": 0.06812429428100586, "reward": 0.9253357648849487, "reward_std": 0.06812427937984467, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.039003822952508926, "sampling/sampling_logp_difference/max": 1.4666242599487305, "sampling/importance_sampling_ratio/min": 0.2307029664516449, "sampling/importance_sampling_ratio/mean": 1.0046266317367554, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.19998472929000854, "clip_ratio/low_mean": 0.007575757801532745, "clip_ratio/low_min": 0.007575757801532745, "clip_ratio/high_mean": 0.0361340157687664, "clip_ratio/high_max": 0.0361340157687664, "clip_ratio/region_mean": 0.04370977357029915, "reward_total_mean": 0.9253357648849487, "reward_meter_mean": 0.9489889144897461, "reward_meter_std": 0.003239382989704609, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9253357648849487, "reward_total_composite_std": 0.06812429428100586} {"timestamp_utc": "2026-04-12T02:50:00Z", "mode": "train", "global_step": 2840, "epoch": 0.11406996826926939, "loss": 0.001, "grad_norm": 0.9888406991958618, "learning_rate": 1.3969696969696972e-06, "num_tokens": 6446736.0, "completions/mean_length": 98.125, "completions/min_length": 98.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.125, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.9993867874145508, "rewards/meter/std": 5.2988518291385844e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993867874145508, "rewards/total_composite/std": 5.2988518291385844e-05, "reward": 0.9993867874145508, "reward_std": 5.3000439947936684e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009926079772412777, "sampling/sampling_logp_difference/max": 1.117924690246582, "sampling/importance_sampling_ratio/min": 0.326957643032074, "sampling/importance_sampling_ratio/mean": 1.0002104043960571, "sampling/importance_sampling_ratio/max": 1.6428247690200806, "entropy": 0.04481238219887018, "clip_ratio/low_mean": 0.0038265305338427424, "clip_ratio/low_min": 0.0038265305338427424, "clip_ratio/high_mean": 0.005076272878795862, "clip_ratio/high_max": 0.005076272878795862, "clip_ratio/region_mean": 0.008902803412638605, "reward_total_mean": 0.9993867874145508, "reward_meter_mean": 0.9993867874145508, "reward_meter_std": 5.2988518291385844e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993867874145508, "reward_total_composite_std": 5.2988518291385844e-05} {"timestamp_utc": "2026-04-12T02:50:05Z", "mode": "train", "global_step": 2841, "epoch": 0.11411013375105435, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.3939393939393942e-06, "num_tokens": 6448576.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00023380780476145446, "sampling/sampling_logp_difference/max": 0.0026471014134585857, "sampling/importance_sampling_ratio/min": 0.9991210103034973, "sampling/importance_sampling_ratio/mean": 1.000229001045227, "sampling/importance_sampling_ratio/max": 1.0026506185531616, "entropy": 0.001806725820642896, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:50:09Z", "mode": "train", "global_step": 2842, "epoch": 0.1141502992328393, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.390909090909091e-06, "num_tokens": 6450400.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00023029858130030334, "sampling/sampling_logp_difference/max": 0.003564245533198118, "sampling/importance_sampling_ratio/min": 0.9999062418937683, "sampling/importance_sampling_ratio/mean": 1.0002295970916748, "sampling/importance_sampling_ratio/max": 1.0035706758499146, "entropy": 0.002088540146360174, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:50:14Z", "mode": "train", "global_step": 2843, "epoch": 0.11419046471462425, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.3878787878787881e-06, "num_tokens": 6452232.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002246771182399243, "sampling/sampling_logp_difference/max": 0.0022279510740190744, "sampling/importance_sampling_ratio/min": 0.999897837638855, "sampling/importance_sampling_ratio/mean": 1.0002238750457764, "sampling/importance_sampling_ratio/max": 1.0022305250167847, "entropy": 0.0019821640598820522, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:50:18Z", "mode": "train", "global_step": 2844, "epoch": 0.11423063019640921, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.384848484848485e-06, "num_tokens": 6453936.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00025937738246284425, "sampling/sampling_logp_difference/max": 0.003465580753982067, "sampling/importance_sampling_ratio/min": 0.9998733997344971, "sampling/importance_sampling_ratio/mean": 1.0002586841583252, "sampling/importance_sampling_ratio/max": 1.0034716129302979, "entropy": 0.0018740687082754448, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:50:22Z", "mode": "train", "global_step": 2845, "epoch": 0.11427079567819416, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.381818181818182e-06, "num_tokens": 6455448.0, "completions/mean_length": 35.0, "completions/min_length": 35.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9980231523513794, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980231523513794, "rewards/total_composite/std": 0.0, "reward": 0.9980231523513794, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0005941917770542204, "sampling/sampling_logp_difference/max": 0.01501617580652237, "sampling/importance_sampling_ratio/min": 0.9850959777832031, "sampling/importance_sampling_ratio/mean": 1.000425934791565, "sampling/importance_sampling_ratio/max": 1.008884072303772, "entropy": 0.004738863033708185, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9980231523513794, "reward_meter_mean": 0.9980231523513794, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9980231523513794, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:50:28Z", "mode": "train", "global_step": 2846, "epoch": 0.11431096115997912, "loss": 0.0054, "grad_norm": 2.306335926055908, "learning_rate": 1.3787878787878788e-06, "num_tokens": 6458571.0, "completions/mean_length": 178.375, "completions/min_length": 178.0, "completions/max_length": 179.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 178.375, "completions/min_terminated_length": 178.0, "completions/max_terminated_length": 179.0, "rewards/meter/mean": 0.9989786148071289, "rewards/meter/std": 0.00010886832023970783, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9444444179534912, "rewards/repeat_penalty/std": 0.059391383081674576, "rewards/total_composite/mean": 0.9434767961502075, "rewards/total_composite/std": 0.05927610024809837, "reward": 0.9434767961502075, "reward_std": 0.05927610024809837, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01770571805536747, "sampling/sampling_logp_difference/max": 1.0042657852172852, "sampling/importance_sampling_ratio/min": 0.36631348729133606, "sampling/importance_sampling_ratio/mean": 1.0038070678710938, "sampling/importance_sampling_ratio/max": 1.79298734664917, "entropy": 0.18307929299771786, "clip_ratio/low_mean": 0.007716872845776379, "clip_ratio/low_min": 0.007716872845776379, "clip_ratio/high_mean": 0.008395581040531397, "clip_ratio/high_max": 0.008395581040531397, "clip_ratio/region_mean": 0.016112453886307776, "reward_total_mean": 0.9434767961502075, "reward_meter_mean": 0.9989786148071289, "reward_meter_std": 0.00010886832023970783, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9444444179534912, "reward_repeat_penalty_std": 0.059391383081674576, "reward_total_composite_mean": 0.9434767961502075, "reward_total_composite_std": 0.05927610024809837} {"timestamp_utc": "2026-04-12T02:50:33Z", "mode": "train", "global_step": 2847, "epoch": 0.11435112664176407, "loss": -0.0027, "grad_norm": 2.3205626010894775, "learning_rate": 1.3757575757575759e-06, "num_tokens": 6460630.0, "completions/mean_length": 80.375, "completions/min_length": 79.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.375, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.9975742101669312, "rewards/meter/std": 0.002095119096338749, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9975742101669312, "rewards/total_composite/std": 0.002095119096338749, "reward": 0.9975742101669312, "reward_std": 0.0020951295737177134, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.034449756145477295, "sampling/sampling_logp_difference/max": 1.1230812072753906, "sampling/importance_sampling_ratio/min": 0.3252760171890259, "sampling/importance_sampling_ratio/mean": 1.0142433643341064, "sampling/importance_sampling_ratio/max": 1.7717058658599854, "entropy": 0.2891568746417761, "clip_ratio/low_mean": 0.01566602219827473, "clip_ratio/low_min": 0.01566602219827473, "clip_ratio/high_mean": 0.016998398234136403, "clip_ratio/high_max": 0.016998398234136403, "clip_ratio/region_mean": 0.032664420432411134, "reward_total_mean": 0.9975742101669312, "reward_meter_mean": 0.9975742101669312, "reward_meter_std": 0.002095119096338749, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9975742101669312, "reward_total_composite_std": 0.002095119096338749} {"timestamp_utc": "2026-04-12T02:50:37Z", "mode": "train", "global_step": 2848, "epoch": 0.11439129212354902, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.3727272727272727e-06, "num_tokens": 6462110.0, "completions/mean_length": 29.0, "completions/min_length": 29.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9957436919212341, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957436919212341, "rewards/total_composite/std": 0.0, "reward": 0.9957436919212341, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0006461237207986414, "sampling/sampling_logp_difference/max": 0.029664982110261917, "sampling/importance_sampling_ratio/min": 0.9707706570625305, "sampling/importance_sampling_ratio/mean": 1.0002989768981934, "sampling/importance_sampling_ratio/max": 1.017333745956421, "entropy": 0.005831975577166304, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9957436919212341, "reward_meter_mean": 0.9957436919212341, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9957436919212341, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:50:42Z", "mode": "train", "global_step": 2849, "epoch": 0.11443145760533398, "loss": -0.0004, "grad_norm": 2.436666965484619, "learning_rate": 1.3696969696969697e-06, "num_tokens": 6464768.0, "completions/mean_length": 142.25, "completions/min_length": 140.0, "completions/max_length": 143.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 142.25, "completions/min_terminated_length": 140.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.9991288781166077, "rewards/meter/std": 0.0001261577708646655, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8928571939468384, "rewards/repeat_penalty/std": 0.10101524740457535, "rewards/total_composite/mean": 0.8920784592628479, "rewards/total_composite/std": 0.10092534869909286, "reward": 0.8920784592628479, "reward_std": 0.10092535614967346, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019777992740273476, "sampling/sampling_logp_difference/max": 1.4602470397949219, "sampling/importance_sampling_ratio/min": 0.2321789264678955, "sampling/importance_sampling_ratio/mean": 1.0004011392593384, "sampling/importance_sampling_ratio/max": 1.7561588287353516, "entropy": 0.15198146365582943, "clip_ratio/low_mean": 0.005313260713592172, "clip_ratio/low_min": 0.005313260713592172, "clip_ratio/high_mean": 0.0078856002073735, "clip_ratio/high_max": 0.0078856002073735, "clip_ratio/region_mean": 0.013198860920965672, "reward_total_mean": 0.8920784592628479, "reward_meter_mean": 0.9991288781166077, "reward_meter_std": 0.0001261577708646655, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8928571939468384, "reward_repeat_penalty_std": 0.10101524740457535, "reward_total_composite_mean": 0.8920784592628479, "reward_total_composite_std": 0.10092534869909286} {"timestamp_utc": "2026-04-12T02:50:46Z", "mode": "train", "global_step": 2850, "epoch": 0.11447162308711893, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.3666666666666668e-06, "num_tokens": 6466328.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002724671212490648, "sampling/sampling_logp_difference/max": 0.0029080850072205067, "sampling/importance_sampling_ratio/min": 0.999262273311615, "sampling/importance_sampling_ratio/mean": 1.0002659559249878, "sampling/importance_sampling_ratio/max": 1.0029124021530151, "entropy": 0.002160783580620773, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:52:04Z", "mode": "eval", "global_step": 2850, "epoch": 0.11447162308711893, "eval_loss": NaN, "eval_runtime": 77.18, "eval_samples_per_second": 1.347, "eval_steps_per_second": 0.168, "eval_num_tokens": 6466328.0, "eval_completions/mean_length": 211.08653846153845, "eval_completions/min_length": 59.92307692307692, "eval_completions/max_length": 411.38461538461536, "eval_completions/clipped_ratio": 0.028846153846153848, "eval_completions/mean_terminated_length": 201.30357360839844, "eval_completions/min_terminated_length": 59.92307692307692, "eval_completions/max_terminated_length": 386.0769230769231, "eval_rewards/meter/mean": 0.7710461295568026, "eval_rewards/meter/std": 0.3627813183344327, "eval_rewards/count_adherence/mean": 0.9523772459763747, "eval_rewards/count_adherence/std": 0.06613255492769755, "eval_rewards/arabic_clean/mean": 0.9615384615384616, "eval_rewards/arabic_clean/std": 0.10878565678229699, "eval_rewards/repeat_penalty/mean": 0.9531235878284161, "eval_rewards/repeat_penalty/std": 0.07651243057961647, "eval_rewards/total_composite/mean": 0.6904892875598028, "eval_rewards/total_composite/std": 0.3804844663693355, "eval_reward": 0.6904892875598028, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.03192363708065106, "eval_sampling/sampling_logp_difference/max": 1.119479619539701, "eval_sampling/importance_sampling_ratio/min": 0.33661178098275113, "eval_sampling/importance_sampling_ratio/mean": 1.009327219082759, "eval_sampling/importance_sampling_ratio/max": 1.5361342888612013, "eval_entropy": 0.3772245920621432, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6904892875598028, "eval_reward_meter_mean": 0.7710461295568026, "eval_reward_meter_std": 0.3627813183344327, "eval_reward_count_adherence_mean": 0.9523772459763747, "eval_reward_count_adherence_std": 0.06613255492769755, "eval_reward_arabic_clean_mean": 0.9615384615384616, "eval_reward_arabic_clean_std": 0.10878565678229699, "eval_reward_repeat_penalty_mean": 0.9531235878284161, "eval_reward_repeat_penalty_std": 0.07651243057961647, "eval_reward_total_composite_mean": 0.6904892875598028, "eval_reward_total_composite_std": 0.3804844663693355} {"timestamp_utc": "2026-04-12T02:52:14Z", "mode": "train", "global_step": 2851, "epoch": 0.11451178856890389, "loss": 0.0037, "grad_norm": 1.7204653024673462, "learning_rate": 1.3636363636363636e-06, "num_tokens": 6469974.0, "completions/mean_length": 248.75, "completions/min_length": 248.0, "completions/max_length": 250.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 248.75, "completions/min_terminated_length": 248.0, "completions/max_terminated_length": 250.0, "rewards/meter/mean": 0.9985492825508118, "rewards/meter/std": 0.000639773381408304, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9230769276618958, "rewards/repeat_penalty/std": 0.07121692597866058, "rewards/total_composite/mean": 0.9217178821563721, "rewards/total_composite/std": 0.0708138644695282, "reward": 0.9217178821563721, "reward_std": 0.070813849568367, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024417459964752197, "sampling/sampling_logp_difference/max": 1.4727954864501953, "sampling/importance_sampling_ratio/min": 0.2292836457490921, "sampling/importance_sampling_ratio/mean": 1.0041165351867676, "sampling/importance_sampling_ratio/max": 1.587019920349121, "entropy": 0.24152424931526184, "clip_ratio/low_mean": 0.004026185255497694, "clip_ratio/low_min": 0.004026185255497694, "clip_ratio/high_mean": 0.010042233159765601, "clip_ratio/high_max": 0.010042233159765601, "clip_ratio/region_mean": 0.014068418415263295, "reward_total_mean": 0.9217178821563721, "reward_meter_mean": 0.9985492825508118, "reward_meter_std": 0.000639773381408304, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9230769276618958, "reward_repeat_penalty_std": 0.07121692597866058, "reward_total_composite_mean": 0.9217178821563721, "reward_total_composite_std": 0.0708138644695282} {"timestamp_utc": "2026-04-12T02:52:19Z", "mode": "train", "global_step": 2852, "epoch": 0.11455195405068884, "loss": -0.0252, "grad_norm": 8.037017822265625, "learning_rate": 1.3606060606060607e-06, "num_tokens": 6471754.0, "completions/mean_length": 65.5, "completions/min_length": 60.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.5, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.951325535774231, "rewards/meter/std": 0.1125522255897522, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.951325535774231, "rewards/total_composite/std": 0.1125522255897522, "reward": 0.951325535774231, "reward_std": 0.112552210688591, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.042092662304639816, "sampling/sampling_logp_difference/max": 1.1297359466552734, "sampling/importance_sampling_ratio/min": 0.3231185972690582, "sampling/importance_sampling_ratio/mean": 1.0115268230438232, "sampling/importance_sampling_ratio/max": 1.7486579418182373, "entropy": 0.38093627616763115, "clip_ratio/low_mean": 0.008333333767950535, "clip_ratio/low_min": 0.008333333767950535, "clip_ratio/high_mean": 0.014985310845077038, "clip_ratio/high_max": 0.014985310845077038, "clip_ratio/region_mean": 0.023318644613027573, "reward_total_mean": 0.951325535774231, "reward_meter_mean": 0.951325535774231, "reward_meter_std": 0.1125522255897522, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.951325535774231, "reward_total_composite_std": 0.1125522255897522} {"timestamp_utc": "2026-04-12T02:52:23Z", "mode": "train", "global_step": 2853, "epoch": 0.1145921195324738, "loss": 0.0008, "grad_norm": 2.253185510635376, "learning_rate": 1.357575757575758e-06, "num_tokens": 6473577.0, "completions/mean_length": 79.875, "completions/min_length": 79.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.875, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9988980293273926, "rewards/meter/std": 0.0002794552710838616, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9988980293273926, "rewards/total_composite/std": 0.0002794552710838616, "reward": 0.9988980293273926, "reward_std": 0.0002794554748106748, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027408134192228317, "sampling/sampling_logp_difference/max": 0.9639253616333008, "sampling/importance_sampling_ratio/min": 0.38139283657073975, "sampling/importance_sampling_ratio/mean": 1.0052258968353271, "sampling/importance_sampling_ratio/max": 1.6166248321533203, "entropy": 0.2539651710540056, "clip_ratio/low_mean": 0.004707278567366302, "clip_ratio/low_min": 0.004707278567366302, "clip_ratio/high_mean": 0.0204123689327389, "clip_ratio/high_max": 0.0204123689327389, "clip_ratio/region_mean": 0.025119647500105202, "reward_total_mean": 0.9988980293273926, "reward_meter_mean": 0.9988980293273926, "reward_meter_std": 0.0002794552710838616, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9988980293273926, "reward_total_composite_std": 0.0002794552710838616} {"timestamp_utc": "2026-04-12T02:52:29Z", "mode": "train", "global_step": 2854, "epoch": 0.11463228501425875, "loss": -0.0055, "grad_norm": 4.1874613761901855, "learning_rate": 1.3545454545454547e-06, "num_tokens": 6475980.0, "completions/mean_length": 126.375, "completions/min_length": 124.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 126.375, "completions/min_terminated_length": 124.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9470775127410889, "rewards/meter/std": 0.12438784539699554, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.929327130317688, "rewards/total_composite/std": 0.12683995068073273, "reward": 0.929327130317688, "reward_std": 0.12683993577957153, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0373864509165287, "sampling/sampling_logp_difference/max": 1.086287498474121, "sampling/importance_sampling_ratio/min": 0.3374670445919037, "sampling/importance_sampling_ratio/mean": 1.0064387321472168, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.34479158371686935, "clip_ratio/low_mean": 0.004032257944345474, "clip_ratio/low_min": 0.004032257944345474, "clip_ratio/high_mean": 0.018657967331819236, "clip_ratio/high_max": 0.018657967331819236, "clip_ratio/region_mean": 0.02269022527616471, "reward_total_mean": 0.929327130317688, "reward_meter_mean": 0.9470775127410889, "reward_meter_std": 0.12438784539699554, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.929327130317688, "reward_total_composite_std": 0.12683995068073273} {"timestamp_utc": "2026-04-12T02:52:33Z", "mode": "train", "global_step": 2855, "epoch": 0.1146724504960437, "loss": 0.0057, "grad_norm": 5.475809097290039, "learning_rate": 1.3515151515151518e-06, "num_tokens": 6477692.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9972702860832214, "rewards/meter/std": 0.00018694026221055537, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9972702860832214, "rewards/total_composite/std": 0.00018694026221055537, "reward": 0.9972702860832214, "reward_std": 0.0001869339175755158, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01253997441381216, "sampling/sampling_logp_difference/max": 0.8330123424530029, "sampling/importance_sampling_ratio/min": 0.4347377419471741, "sampling/importance_sampling_ratio/mean": 1.0026929378509521, "sampling/importance_sampling_ratio/max": 1.7445498704910278, "entropy": 0.06320370081812143, "clip_ratio/low_mean": 0.0020491802133619785, "clip_ratio/low_min": 0.0020491802133619785, "clip_ratio/high_mean": 0.006147540640085936, "clip_ratio/high_max": 0.006147540640085936, "clip_ratio/region_mean": 0.008196720853447914, "reward_total_mean": 0.9972702860832214, "reward_meter_mean": 0.9972702860832214, "reward_meter_std": 0.00018694026221055537, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9972702860832214, "reward_total_composite_std": 0.00018694026221055537} {"timestamp_utc": "2026-04-12T02:52:38Z", "mode": "train", "global_step": 2856, "epoch": 0.11471261597782866, "loss": -0.0073, "grad_norm": 3.800494909286499, "learning_rate": 1.3484848484848486e-06, "num_tokens": 6479747.0, "completions/mean_length": 100.875, "completions/min_length": 98.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.875, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.9850212931632996, "rewards/meter/std": 0.02900601737201214, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9850212931632996, "rewards/total_composite/std": 0.02900601737201214, "reward": 0.9850212931632996, "reward_std": 0.02900601364672184, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031728874891996384, "sampling/sampling_logp_difference/max": 1.3605375289916992, "sampling/importance_sampling_ratio/min": 0.25652286410331726, "sampling/importance_sampling_ratio/mean": 1.0072658061981201, "sampling/importance_sampling_ratio/max": 1.8992688655853271, "entropy": 0.2935164403170347, "clip_ratio/low_mean": 0.0012755101779475808, "clip_ratio/low_min": 0.0012755101779475808, "clip_ratio/high_mean": 0.02479799627326429, "clip_ratio/high_max": 0.02479799627326429, "clip_ratio/region_mean": 0.02607350645121187, "reward_total_mean": 0.9850212931632996, "reward_meter_mean": 0.9850212931632996, "reward_meter_std": 0.02900601737201214, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9850212931632996, "reward_total_composite_std": 0.02900601737201214} {"timestamp_utc": "2026-04-12T02:52:42Z", "mode": "train", "global_step": 2857, "epoch": 0.11475278145961361, "loss": 0.0001, "grad_norm": 0.22853942215442657, "learning_rate": 1.3454545454545457e-06, "num_tokens": 6481515.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9981458783149719, "rewards/meter/std": 1.3051687346887775e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981458783149719, "rewards/total_composite/std": 1.3051687346887775e-05, "reward": 0.9981458783149719, "reward_std": 1.3051648238615599e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005550025030970573, "sampling/sampling_logp_difference/max": 0.3850884437561035, "sampling/importance_sampling_ratio/min": 0.8009796142578125, "sampling/importance_sampling_ratio/mean": 1.002258539199829, "sampling/importance_sampling_ratio/max": 1.469744324684143, "entropy": 0.053996546659618616, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.009469697251915932, "clip_ratio/high_max": 0.009469697251915932, "clip_ratio/region_mean": 0.009469697251915932, "reward_total_mean": 0.9981458783149719, "reward_meter_mean": 0.9981458783149719, "reward_meter_std": 1.3051687346887775e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981458783149719, "reward_total_composite_std": 1.3051687346887775e-05} {"timestamp_utc": "2026-04-12T02:52:50Z", "mode": "train", "global_step": 2858, "epoch": 0.11479294694139856, "loss": 0.0307, "grad_norm": 2.1501612663269043, "learning_rate": 1.3424242424242425e-06, "num_tokens": 6485258.0, "completions/mean_length": 279.875, "completions/min_length": 261.0, "completions/max_length": 298.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 279.875, "completions/min_terminated_length": 261.0, "completions/max_terminated_length": 298.0, "rewards/meter/mean": 0.9832162857055664, "rewards/meter/std": 0.0437212772667408, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.06681530922651291, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8965686559677124, "rewards/repeat_penalty/std": 0.05159881338477135, "rewards/total_composite/mean": 0.8241852521896362, "rewards/total_composite/std": 0.05549175664782524, "reward": 0.8241852521896362, "reward_std": 0.05549176037311554, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03513329103589058, "sampling/sampling_logp_difference/max": 2.449089527130127, "sampling/importance_sampling_ratio/min": 0.08637219667434692, "sampling/importance_sampling_ratio/mean": 1.0033777952194214, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2756977826356888, "clip_ratio/low_mean": 0.014873409294523299, "clip_ratio/low_min": 0.014873409294523299, "clip_ratio/high_mean": 0.0062042842619121075, "clip_ratio/high_max": 0.0062042842619121075, "clip_ratio/region_mean": 0.021077693556435406, "reward_total_mean": 0.8241852521896362, "reward_meter_mean": 0.9832162857055664, "reward_meter_std": 0.0437212772667408, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.06681530922651291, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8965686559677124, "reward_repeat_penalty_std": 0.05159881338477135, "reward_total_composite_mean": 0.8241852521896362, "reward_total_composite_std": 0.05549175664782524} {"timestamp_utc": "2026-04-12T02:52:54Z", "mode": "train", "global_step": 2859, "epoch": 0.11483311242318352, "loss": -0.0101, "grad_norm": 3.014261245727539, "learning_rate": 1.3393939393939395e-06, "num_tokens": 6487078.0, "completions/mean_length": 66.5, "completions/min_length": 65.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.5, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9922887086868286, "rewards/meter/std": 0.0029313687700778246, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9922887086868286, "rewards/total_composite/std": 0.0029313687700778246, "reward": 0.9922887086868286, "reward_std": 0.0029313701670616865, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031141582876443863, "sampling/sampling_logp_difference/max": 1.0435974597930908, "sampling/importance_sampling_ratio/min": 0.3521854281425476, "sampling/importance_sampling_ratio/mean": 1.0089040994644165, "sampling/importance_sampling_ratio/max": 1.5588260889053345, "entropy": 0.23347646743059158, "clip_ratio/low_mean": 0.009471436496824026, "clip_ratio/low_min": 0.009471436496824026, "clip_ratio/high_mean": 0.01487299520522356, "clip_ratio/high_max": 0.01487299520522356, "clip_ratio/region_mean": 0.024344431702047586, "reward_total_mean": 0.9922887086868286, "reward_meter_mean": 0.9922887086868286, "reward_meter_std": 0.0029313687700778246, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9922887086868286, "reward_total_composite_std": 0.0029313687700778246} {"timestamp_utc": "2026-04-12T02:52:59Z", "mode": "train", "global_step": 2860, "epoch": 0.11487327790496847, "loss": -0.0013, "grad_norm": 0.3336952030658722, "learning_rate": 1.3363636363636364e-06, "num_tokens": 6488807.0, "completions/mean_length": 66.125, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.125, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981107711791992, "rewards/meter/std": 5.631321982946247e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981107711791992, "rewards/total_composite/std": 5.631321982946247e-05, "reward": 0.9981107711791992, "reward_std": 5.631066233036108e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007993371225893497, "sampling/sampling_logp_difference/max": 0.5051336288452148, "sampling/importance_sampling_ratio/min": 0.6034249067306519, "sampling/importance_sampling_ratio/mean": 1.004895567893982, "sampling/importance_sampling_ratio/max": 1.383058786392212, "entropy": 0.05899450369179249, "clip_ratio/low_mean": 0.011363636702299118, "clip_ratio/low_min": 0.011363636702299118, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.011363636702299118, "reward_total_mean": 0.9981107711791992, "reward_meter_mean": 0.9981107711791992, "reward_meter_std": 5.631321982946247e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981107711791992, "reward_total_composite_std": 5.631321982946247e-05} {"timestamp_utc": "2026-04-12T02:53:02Z", "mode": "train", "global_step": 2861, "epoch": 0.11491344338675342, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.3333333333333334e-06, "num_tokens": 6490287.0, "completions/mean_length": 32.0, "completions/min_length": 32.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "reward": 0.9992982149124146, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 3.3981952583417296e-05, "sampling/sampling_logp_difference/max": 0.0006292310426943004, "sampling/importance_sampling_ratio/min": 0.9998278617858887, "sampling/importance_sampling_ratio/mean": 1.0000324249267578, "sampling/importance_sampling_ratio/max": 1.0006294250488281, "entropy": 0.0002626157729537226, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9992982149124146, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:53:09Z", "mode": "train", "global_step": 2862, "epoch": 0.11495360886853838, "loss": -0.0358, "grad_norm": 3.7277348041534424, "learning_rate": 1.3303030303030305e-06, "num_tokens": 6493447.0, "completions/mean_length": 195.0, "completions/min_length": 173.0, "completions/max_length": 207.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 195.0, "completions/min_terminated_length": 173.0, "completions/max_terminated_length": 207.0, "rewards/meter/mean": 0.9989534616470337, "rewards/meter/std": 0.00048265265650115907, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.07715168595314026, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9573421478271484, "rewards/total_composite/std": 0.07722419500350952, "reward": 0.9573421478271484, "reward_std": 0.07722420245409012, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05510193854570389, "sampling/sampling_logp_difference/max": 1.9347803592681885, "sampling/importance_sampling_ratio/min": 0.14445599913597107, "sampling/importance_sampling_ratio/mean": 1.0075865983963013, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4370429702103138, "clip_ratio/low_mean": 0.01037695212289691, "clip_ratio/low_min": 0.01037695212289691, "clip_ratio/high_mean": 0.040064468048512936, "clip_ratio/high_max": 0.040064468048512936, "clip_ratio/region_mean": 0.050441420171409845, "reward_total_mean": 0.9573421478271484, "reward_meter_mean": 0.9989534616470337, "reward_meter_std": 0.00048265265650115907, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.07715168595314026, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9573421478271484, "reward_total_composite_std": 0.07722419500350952} {"timestamp_utc": "2026-04-12T02:53:13Z", "mode": "train", "global_step": 2863, "epoch": 0.11499377435032333, "loss": 0.0, "grad_norm": 0.6136480569839478, "learning_rate": 1.3272727272727273e-06, "num_tokens": 6495258.0, "completions/mean_length": 71.375, "completions/min_length": 71.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.375, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9994188547134399, "rewards/meter/std": 3.3046217140508816e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994188547134399, "rewards/total_composite/std": 3.3046217140508816e-05, "reward": 0.9994188547134399, "reward_std": 3.304608617327176e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01137758232653141, "sampling/sampling_logp_difference/max": 1.0333290100097656, "sampling/importance_sampling_ratio/min": 0.3558204770088196, "sampling/importance_sampling_ratio/mean": 1.0004888772964478, "sampling/importance_sampling_ratio/max": 1.2199125289916992, "entropy": 0.07642534840852022, "clip_ratio/low_mean": 0.005208333372138441, "clip_ratio/low_min": 0.005208333372138441, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.005208333372138441, "reward_total_mean": 0.9994188547134399, "reward_meter_mean": 0.9994188547134399, "reward_meter_std": 3.3046217140508816e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994188547134399, "reward_total_composite_std": 3.3046217140508816e-05} {"timestamp_utc": "2026-04-12T02:53:19Z", "mode": "train", "global_step": 2864, "epoch": 0.11503393983210829, "loss": 0.0017, "grad_norm": 2.3443167209625244, "learning_rate": 1.3242424242424243e-06, "num_tokens": 6498081.0, "completions/mean_length": 164.875, "completions/min_length": 162.0, "completions/max_length": 167.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 164.875, "completions/min_terminated_length": 162.0, "completions/max_terminated_length": 167.0, "rewards/meter/mean": 0.9993668794631958, "rewards/meter/std": 0.0001238392578670755, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9861111044883728, "rewards/repeat_penalty/std": 0.03928370773792267, "rewards/total_composite/mean": 0.9854899644851685, "rewards/total_composite/std": 0.03934282064437866, "reward": 0.9854899644851685, "reward_std": 0.039342816919088364, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024905560538172722, "sampling/sampling_logp_difference/max": 1.7190942764282227, "sampling/importance_sampling_ratio/min": 0.17922841012477875, "sampling/importance_sampling_ratio/mean": 1.000923991203308, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15149082150310278, "clip_ratio/low_mean": 0.0030303029343485832, "clip_ratio/low_min": 0.0030303029343485832, "clip_ratio/high_mean": 0.013674213085323572, "clip_ratio/high_max": 0.013674213085323572, "clip_ratio/region_mean": 0.016704516019672155, "reward_total_mean": 0.9854899644851685, "reward_meter_mean": 0.9993668794631958, "reward_meter_std": 0.0001238392578670755, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9861111044883728, "reward_repeat_penalty_std": 0.03928370773792267, "reward_total_composite_mean": 0.9854899644851685, "reward_total_composite_std": 0.03934282064437866} {"timestamp_utc": "2026-04-12T02:53:28Z", "mode": "train", "global_step": 2865, "epoch": 0.11507410531389324, "loss": -0.0115, "grad_norm": 1.4304178953170776, "learning_rate": 1.3212121212121212e-06, "num_tokens": 6503361.0, "completions/mean_length": 442.0, "completions/min_length": 430.0, "completions/max_length": 470.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 442.0, "completions/min_terminated_length": 430.0, "completions/max_terminated_length": 470.0, "rewards/meter/mean": 0.9988847970962524, "rewards/meter/std": 0.0004295332182664424, "rewards/count_adherence/mean": 0.8035714030265808, "rewards/count_adherence/std": 0.03306501731276512, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9593685269355774, "rewards/repeat_penalty/std": 0.030346577987074852, "rewards/total_composite/mean": 0.7700097560882568, "rewards/total_composite/std": 0.03843219205737114, "reward": 0.7700097560882568, "reward_std": 0.03843218460679054, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05211654305458069, "sampling/sampling_logp_difference/max": 1.5292320251464844, "sampling/importance_sampling_ratio/min": 0.21670202910900116, "sampling/importance_sampling_ratio/mean": 1.010416030883789, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4746365137398243, "clip_ratio/low_mean": 0.008053920813836157, "clip_ratio/low_min": 0.008053920813836157, "clip_ratio/high_mean": 0.019567076116800308, "clip_ratio/high_max": 0.019567076116800308, "clip_ratio/region_mean": 0.027620996930636466, "reward_total_mean": 0.7700097560882568, "reward_meter_mean": 0.9988847970962524, "reward_meter_std": 0.0004295332182664424, "reward_count_adherence_mean": 0.8035714030265808, "reward_count_adherence_std": 0.03306501731276512, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9593685269355774, "reward_repeat_penalty_std": 0.030346577987074852, "reward_total_composite_mean": 0.7700097560882568, "reward_total_composite_std": 0.03843219205737114} {"timestamp_utc": "2026-04-12T02:53:32Z", "mode": "train", "global_step": 2866, "epoch": 0.1151142707956782, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.3181818181818182e-06, "num_tokens": 6504873.0, "completions/mean_length": 35.0, "completions/min_length": 35.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9980231523513794, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980231523513794, "rewards/total_composite/std": 0.0, "reward": 0.9980231523513794, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0006467251223511994, "sampling/sampling_logp_difference/max": 0.01518770307302475, "sampling/importance_sampling_ratio/min": 0.9985290169715881, "sampling/importance_sampling_ratio/mean": 1.0006215572357178, "sampling/importance_sampling_ratio/max": 1.015303611755371, "entropy": 0.0056428477400913835, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9980231523513794, "reward_meter_mean": 0.9980231523513794, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9980231523513794, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:53:37Z", "mode": "train", "global_step": 2867, "epoch": 0.11515443627746315, "loss": 0.005, "grad_norm": 4.305418968200684, "learning_rate": 1.315151515151515e-06, "num_tokens": 6506642.0, "completions/mean_length": 64.125, "completions/min_length": 64.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.125, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9991264343261719, "rewards/meter/std": 0.0007721302099525928, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991264343261719, "rewards/total_composite/std": 0.0007721302099525928, "reward": 0.9991264343261719, "reward_std": 0.0007721302681602538, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00290035386569798, "sampling/sampling_logp_difference/max": 1.102513313293457, "sampling/importance_sampling_ratio/min": 0.33203551173210144, "sampling/importance_sampling_ratio/mean": 0.9988252520561218, "sampling/importance_sampling_ratio/max": 1.0192989110946655, "entropy": 0.0063372578588314354, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9991264343261719, "reward_meter_mean": 0.9991264343261719, "reward_meter_std": 0.0007721302099525928, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9991264343261719, "reward_total_composite_std": 0.0007721302099525928} {"timestamp_utc": "2026-04-12T02:53:41Z", "mode": "train", "global_step": 2868, "epoch": 0.1151946017592481, "loss": -0.0079, "grad_norm": 2.3818113803863525, "learning_rate": 1.3121212121212123e-06, "num_tokens": 6508612.0, "completions/mean_length": 80.25, "completions/min_length": 77.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.25, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9989343285560608, "rewards/meter/std": 0.00022084804368205369, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989343285560608, "rewards/total_composite/std": 0.00022084804368205369, "reward": 0.9989343285560608, "reward_std": 0.00022085075033828616, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023430120199918747, "sampling/sampling_logp_difference/max": 1.1872577667236328, "sampling/importance_sampling_ratio/min": 0.3050566613674164, "sampling/importance_sampling_ratio/mean": 1.0084359645843506, "sampling/importance_sampling_ratio/max": 1.6601855754852295, "entropy": 0.21316692978143692, "clip_ratio/low_mean": 0.003246753243729472, "clip_ratio/low_min": 0.003246753243729472, "clip_ratio/high_mean": 0.006192129687406123, "clip_ratio/high_max": 0.006192129687406123, "clip_ratio/region_mean": 0.009438882931135595, "reward_total_mean": 0.9989343285560608, "reward_meter_mean": 0.9989343285560608, "reward_meter_std": 0.00022084804368205369, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9989343285560608, "reward_total_composite_std": 0.00022084804368205369} {"timestamp_utc": "2026-04-12T02:53:46Z", "mode": "train", "global_step": 2869, "epoch": 0.11523476724103306, "loss": -0.0098, "grad_norm": 2.163902997970581, "learning_rate": 1.3090909090909093e-06, "num_tokens": 6510609.0, "completions/mean_length": 96.625, "completions/min_length": 94.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.625, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9977865219116211, "rewards/meter/std": 0.0006117028533481061, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977865219116211, "rewards/total_composite/std": 0.0006117028533481061, "reward": 0.9977865219116211, "reward_std": 0.0006116984295658767, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016453739255666733, "sampling/sampling_logp_difference/max": 1.150710105895996, "sampling/importance_sampling_ratio/min": 0.3164120018482208, "sampling/importance_sampling_ratio/mean": 1.0011489391326904, "sampling/importance_sampling_ratio/max": 1.840904712677002, "entropy": 0.14486887026578188, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.006416999618522823, "clip_ratio/high_max": 0.006416999618522823, "clip_ratio/region_mean": 0.006416999618522823, "reward_total_mean": 0.9977865219116211, "reward_meter_mean": 0.9977865219116211, "reward_meter_std": 0.0006117028533481061, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9977865219116211, "reward_total_composite_std": 0.0006117028533481061} {"timestamp_utc": "2026-04-12T02:53:50Z", "mode": "train", "global_step": 2870, "epoch": 0.11527493272281801, "loss": -0.0, "grad_norm": 1.1164172887802124, "learning_rate": 1.3060606060606062e-06, "num_tokens": 6512337.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9973324537277222, "rewards/meter/std": 2.97028473141836e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973324537277222, "rewards/total_composite/std": 2.97028473141836e-05, "reward": 0.9973324537277222, "reward_std": 2.9693699616473168e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010174022056162357, "sampling/sampling_logp_difference/max": 0.505859375, "sampling/importance_sampling_ratio/min": 0.6083394289016724, "sampling/importance_sampling_ratio/mean": 1.003736138343811, "sampling/importance_sampling_ratio/max": 1.6584101915359497, "entropy": 0.05785668967291713, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0020491802133619785, "clip_ratio/high_max": 0.0020491802133619785, "clip_ratio/region_mean": 0.0020491802133619785, "reward_total_mean": 0.9973324537277222, "reward_meter_mean": 0.9973324537277222, "reward_meter_std": 2.97028473141836e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9973324537277222, "reward_total_composite_std": 2.97028473141836e-05} {"timestamp_utc": "2026-04-12T02:53:55Z", "mode": "train", "global_step": 2871, "epoch": 0.11531509820460296, "loss": 0.0021, "grad_norm": 1.2812321186065674, "learning_rate": 1.3030303030303032e-06, "num_tokens": 6514415.0, "completions/mean_length": 79.75, "completions/min_length": 78.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.75, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9988850355148315, "rewards/meter/std": 0.00015237655316013843, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9988850355148315, "rewards/total_composite/std": 0.00015237655316013843, "reward": 0.9988850355148315, "reward_std": 0.00015238845662679523, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02608993463218212, "sampling/sampling_logp_difference/max": 0.9019899368286133, "sampling/importance_sampling_ratio/min": 0.4057614207267761, "sampling/importance_sampling_ratio/mean": 1.0047295093536377, "sampling/importance_sampling_ratio/max": 1.7389991283416748, "entropy": 0.22218100167810917, "clip_ratio/low_mean": 0.012681030901148915, "clip_ratio/low_min": 0.012681030901148915, "clip_ratio/high_mean": 0.01095727866049856, "clip_ratio/high_max": 0.01095727866049856, "clip_ratio/region_mean": 0.023638309561647475, "reward_total_mean": 0.9988850355148315, "reward_meter_mean": 0.9988850355148315, "reward_meter_std": 0.00015237655316013843, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9988850355148315, "reward_total_composite_std": 0.00015237655316013843} {"timestamp_utc": "2026-04-12T02:54:02Z", "mode": "train", "global_step": 2872, "epoch": 0.11535526368638792, "loss": 0.0065, "grad_norm": 1.5020183324813843, "learning_rate": 1.3e-06, "num_tokens": 6517818.0, "completions/mean_length": 249.375, "completions/min_length": 248.0, "completions/max_length": 251.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 249.375, "completions/min_terminated_length": 248.0, "completions/max_terminated_length": 251.0, "rewards/meter/mean": 0.9985396862030029, "rewards/meter/std": 0.0008591708028689027, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9519230723381042, "rewards/repeat_penalty/std": 0.05723259598016739, "rewards/total_composite/mean": 0.9505188465118408, "rewards/total_composite/std": 0.05688954144716263, "reward": 0.9505188465118408, "reward_std": 0.05688953027129173, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025731699541211128, "sampling/sampling_logp_difference/max": 1.5636186599731445, "sampling/importance_sampling_ratio/min": 0.20937703549861908, "sampling/importance_sampling_ratio/mean": 1.0057145357131958, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.27019675448536873, "clip_ratio/low_mean": 0.003998056286945939, "clip_ratio/low_min": 0.003998056286945939, "clip_ratio/high_mean": 0.0125664914958179, "clip_ratio/high_max": 0.0125664914958179, "clip_ratio/region_mean": 0.01656454778276384, "reward_total_mean": 0.9505188465118408, "reward_meter_mean": 0.9985396862030029, "reward_meter_std": 0.0008591708028689027, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9519230723381042, "reward_repeat_penalty_std": 0.05723259598016739, "reward_total_composite_mean": 0.9505188465118408, "reward_total_composite_std": 0.05688954144716263} {"timestamp_utc": "2026-04-12T02:54:06Z", "mode": "train", "global_step": 2873, "epoch": 0.11539542916817287, "loss": -0.0005, "grad_norm": 0.2703946828842163, "learning_rate": 1.296969696969697e-06, "num_tokens": 6519782.0, "completions/mean_length": 66.5, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.5, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981412291526794, "rewards/meter/std": 1.5779543900862336e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981412291526794, "rewards/total_composite/std": 1.5779543900862336e-05, "reward": 0.9981412291526794, "reward_std": 1.5772715414641425e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009107197634875774, "sampling/sampling_logp_difference/max": 0.5398116111755371, "sampling/importance_sampling_ratio/min": 0.5828580856323242, "sampling/importance_sampling_ratio/mean": 1.0045442581176758, "sampling/importance_sampling_ratio/max": 1.4784173965454102, "entropy": 0.057810590136796236, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0018656715983524919, "reward_total_mean": 0.9981412291526794, "reward_meter_mean": 0.9981412291526794, "reward_meter_std": 1.5779543900862336e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981412291526794, "reward_total_composite_std": 1.5779543900862336e-05} {"timestamp_utc": "2026-04-12T02:54:10Z", "mode": "train", "global_step": 2874, "epoch": 0.11543559464995783, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.2939393939393941e-06, "num_tokens": 6521271.0, "completions/mean_length": 36.125, "completions/min_length": 36.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.125, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9996045231819153, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9996045231819153, "rewards/total_composite/std": 0.0, "reward": 0.9996045231819153, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0043750000186264515, "sampling/sampling_logp_difference/max": 0.496179461479187, "sampling/importance_sampling_ratio/min": 0.6088523864746094, "sampling/importance_sampling_ratio/mean": 0.9987858533859253, "sampling/importance_sampling_ratio/max": 1.0343594551086426, "entropy": 0.02826439938507974, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9996045231819153, "reward_meter_mean": 0.9996045231819153, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9996045231819153, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:54:15Z", "mode": "train", "global_step": 2875, "epoch": 0.11547576013174278, "loss": 0.0072, "grad_norm": 5.8196916580200195, "learning_rate": 1.290909090909091e-06, "num_tokens": 6523051.0, "completions/mean_length": 66.5, "completions/min_length": 65.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.5, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9894192218780518, "rewards/meter/std": 0.010708401910960674, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9894192218780518, "rewards/total_composite/std": 0.010708401910960674, "reward": 0.9894192218780518, "reward_std": 0.01070839911699295, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02634219266474247, "sampling/sampling_logp_difference/max": 2.222754955291748, "sampling/importance_sampling_ratio/min": 0.10831031203269958, "sampling/importance_sampling_ratio/mean": 1.0059653520584106, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17508152220398188, "clip_ratio/low_mean": 0.0018115942366421223, "clip_ratio/low_min": 0.0018115942366421223, "clip_ratio/high_mean": 0.007520091836340725, "clip_ratio/high_max": 0.007520091836340725, "clip_ratio/region_mean": 0.009331686072982848, "reward_total_mean": 0.9894192218780518, "reward_meter_mean": 0.9894192218780518, "reward_meter_std": 0.010708401910960674, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9894192218780518, "reward_total_composite_std": 0.010708401910960674} {"timestamp_utc": "2026-04-12T02:54:19Z", "mode": "train", "global_step": 2876, "epoch": 0.11551592561352773, "loss": -0.0001, "grad_norm": 0.7274681925773621, "learning_rate": 1.287878787878788e-06, "num_tokens": 6525011.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9973265528678894, "rewards/meter/std": 2.102993312291801e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973265528678894, "rewards/total_composite/std": 2.102993312291801e-05, "reward": 0.9973265528678894, "reward_std": 2.101841710100416e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010243063792586327, "sampling/sampling_logp_difference/max": 0.8772482872009277, "sampling/importance_sampling_ratio/min": 0.41592586040496826, "sampling/importance_sampling_ratio/mean": 0.9998522400856018, "sampling/importance_sampling_ratio/max": 1.5096032619476318, "entropy": 0.05711106630042195, "clip_ratio/low_mean": 0.004098360426723957, "clip_ratio/low_min": 0.004098360426723957, "clip_ratio/high_mean": 0.0020491802133619785, "clip_ratio/high_max": 0.0020491802133619785, "clip_ratio/region_mean": 0.006147540640085936, "reward_total_mean": 0.9973265528678894, "reward_meter_mean": 0.9973265528678894, "reward_meter_std": 2.102993312291801e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9973265528678894, "reward_total_composite_std": 2.102993312291801e-05} {"timestamp_utc": "2026-04-12T02:54:24Z", "mode": "train", "global_step": 2877, "epoch": 0.11555609109531269, "loss": -0.0056, "grad_norm": 2.641002655029297, "learning_rate": 1.2848484848484848e-06, "num_tokens": 6527326.0, "completions/mean_length": 118.375, "completions/min_length": 115.0, "completions/max_length": 123.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.375, "completions/min_terminated_length": 115.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.9990575313568115, "rewards/meter/std": 0.00029957492370158434, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990575313568115, "rewards/total_composite/std": 0.00029957492370158434, "reward": 0.9990575313568115, "reward_std": 0.00029957492370158434, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0404638797044754, "sampling/sampling_logp_difference/max": 1.9206485748291016, "sampling/importance_sampling_ratio/min": 0.14651189744472504, "sampling/importance_sampling_ratio/mean": 1.0039846897125244, "sampling/importance_sampling_ratio/max": 1.9006513357162476, "entropy": 0.35712629184126854, "clip_ratio/low_mean": 0.015123688150197268, "clip_ratio/low_min": 0.015123688150197268, "clip_ratio/high_mean": 0.023920452571474016, "clip_ratio/high_max": 0.023920452571474016, "clip_ratio/region_mean": 0.03904414072167128, "reward_total_mean": 0.9990575313568115, "reward_meter_mean": 0.9990575313568115, "reward_meter_std": 0.00029957492370158434, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990575313568115, "reward_total_composite_std": 0.00029957492370158434} {"timestamp_utc": "2026-04-12T02:54:32Z", "mode": "train", "global_step": 2878, "epoch": 0.11559625657709764, "loss": 0.0099, "grad_norm": 2.2012033462524414, "learning_rate": 1.2818181818181819e-06, "num_tokens": 6531545.0, "completions/mean_length": 302.375, "completions/min_length": 294.0, "completions/max_length": 310.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 302.375, "completions/min_terminated_length": 294.0, "completions/max_terminated_length": 310.0, "rewards/meter/mean": 0.9952223896980286, "rewards/meter/std": 0.0032623247243463993, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9802631139755249, "rewards/repeat_penalty/std": 0.027239417657256126, "rewards/total_composite/mean": 0.975556492805481, "rewards/total_composite/std": 0.02634783275425434, "reward": 0.975556492805481, "reward_std": 0.026347849518060684, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05190505087375641, "sampling/sampling_logp_difference/max": 1.4298324584960938, "sampling/importance_sampling_ratio/min": 0.23934902250766754, "sampling/importance_sampling_ratio/mean": 1.0102460384368896, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47275901213288307, "clip_ratio/low_mean": 0.012780054472386837, "clip_ratio/low_min": 0.012780054472386837, "clip_ratio/high_mean": 0.030017988989129663, "clip_ratio/high_max": 0.030017988989129663, "clip_ratio/region_mean": 0.0427980434615165, "reward_total_mean": 0.975556492805481, "reward_meter_mean": 0.9952223896980286, "reward_meter_std": 0.0032623247243463993, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9802631139755249, "reward_repeat_penalty_std": 0.027239417657256126, "reward_total_composite_mean": 0.975556492805481, "reward_total_composite_std": 0.02634783275425434} {"timestamp_utc": "2026-04-12T02:54:36Z", "mode": "train", "global_step": 2879, "epoch": 0.1156364220588826, "loss": -0.0009, "grad_norm": 0.22587938606739044, "learning_rate": 1.2787878787878787e-06, "num_tokens": 6533695.0, "completions/mean_length": 97.75, "completions/min_length": 97.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.75, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9994171857833862, "rewards/meter/std": 3.359419133630581e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994171857833862, "rewards/total_composite/std": 3.359419133630581e-05, "reward": 0.9994171857833862, "reward_std": 3.360588743817061e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009699815884232521, "sampling/sampling_logp_difference/max": 2.2170257568359375, "sampling/importance_sampling_ratio/min": 0.569092333316803, "sampling/importance_sampling_ratio/mean": 1.0023701190948486, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.04796775570139289, "clip_ratio/low_mean": 0.003839680110104382, "clip_ratio/low_min": 0.003839680110104382, "clip_ratio/high_mean": 0.0025510203558951616, "clip_ratio/high_max": 0.0025510203558951616, "clip_ratio/region_mean": 0.006390700465999544, "reward_total_mean": 0.9994171857833862, "reward_meter_mean": 0.9994171857833862, "reward_meter_std": 3.359419133630581e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994171857833862, "reward_total_composite_std": 3.359419133630581e-05} {"timestamp_utc": "2026-04-12T02:54:42Z", "mode": "train", "global_step": 2880, "epoch": 0.11567658754066755, "loss": 0.0276, "grad_norm": 3.3986284732818604, "learning_rate": 1.2757575757575758e-06, "num_tokens": 6536129.0, "completions/mean_length": 131.25, "completions/min_length": 127.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.25, "completions/min_terminated_length": 127.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.9821168184280396, "rewards/meter/std": 0.010360779240727425, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9821168184280396, "rewards/total_composite/std": 0.010360779240727425, "reward": 0.9821168184280396, "reward_std": 0.010360777378082275, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.045867305248975754, "sampling/sampling_logp_difference/max": 1.2023324966430664, "sampling/importance_sampling_ratio/min": 0.3004924952983856, "sampling/importance_sampling_ratio/mean": 1.0092159509658813, "sampling/importance_sampling_ratio/max": 1.8226743936538696, "entropy": 0.46811263263225555, "clip_ratio/low_mean": 0.014072955353185534, "clip_ratio/low_min": 0.014072955353185534, "clip_ratio/high_mean": 0.01734496164135635, "clip_ratio/high_max": 0.01734496164135635, "clip_ratio/region_mean": 0.03141791699454188, "reward_total_mean": 0.9821168184280396, "reward_meter_mean": 0.9821168184280396, "reward_meter_std": 0.010360779240727425, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9821168184280396, "reward_total_composite_std": 0.010360779240727425} {"timestamp_utc": "2026-04-12T02:54:47Z", "mode": "train", "global_step": 2881, "epoch": 0.1157167530224525, "loss": -0.0002, "grad_norm": 0.9868201613426208, "learning_rate": 1.2727272727272728e-06, "num_tokens": 6538228.0, "completions/mean_length": 97.375, "completions/min_length": 97.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.375, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9979463815689087, "rewards/meter/std": 6.646273686783388e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979463815689087, "rewards/total_composite/std": 6.646273686783388e-05, "reward": 0.9979463815689087, "reward_std": 6.644289533142e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013669819571077824, "sampling/sampling_logp_difference/max": 0.7892913818359375, "sampling/importance_sampling_ratio/min": 0.4541665017604828, "sampling/importance_sampling_ratio/mean": 1.0037821531295776, "sampling/importance_sampling_ratio/max": 1.4697163105010986, "entropy": 0.1293118530884385, "clip_ratio/low_mean": 0.005141489440575242, "clip_ratio/low_min": 0.005141489440575242, "clip_ratio/high_mean": 0.0038528296863660216, "clip_ratio/high_max": 0.0038528296863660216, "clip_ratio/region_mean": 0.008994319126941264, "reward_total_mean": 0.9979463815689087, "reward_meter_mean": 0.9979463815689087, "reward_meter_std": 6.646273686783388e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9979463815689087, "reward_total_composite_std": 6.646273686783388e-05} {"timestamp_utc": "2026-04-12T02:54:51Z", "mode": "train", "global_step": 2882, "epoch": 0.11575691850423746, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.2696969696969698e-06, "num_tokens": 6540020.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00015525527123827487, "sampling/sampling_logp_difference/max": 0.005098958499729633, "sampling/importance_sampling_ratio/min": 0.9974856972694397, "sampling/importance_sampling_ratio/mean": 1.0001249313354492, "sampling/importance_sampling_ratio/max": 1.0051120519638062, "entropy": 0.0014212291716830805, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:54:57Z", "mode": "train", "global_step": 2883, "epoch": 0.11579708398602241, "loss": -0.0014, "grad_norm": 1.7158304452896118, "learning_rate": 1.2666666666666669e-06, "num_tokens": 6542928.0, "completions/mean_length": 142.5, "completions/min_length": 141.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 142.5, "completions/min_terminated_length": 141.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.998982846736908, "rewards/meter/std": 0.00020013147150166333, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998982846736908, "rewards/total_composite/std": 0.00020013147150166333, "reward": 0.998982846736908, "reward_std": 0.00020013845642097294, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022738341242074966, "sampling/sampling_logp_difference/max": 0.9792003631591797, "sampling/importance_sampling_ratio/min": 0.3756113350391388, "sampling/importance_sampling_ratio/mean": 1.0040578842163086, "sampling/importance_sampling_ratio/max": 1.6984710693359375, "entropy": 0.2192006167024374, "clip_ratio/low_mean": 0.004407651489600539, "clip_ratio/low_min": 0.004407651489600539, "clip_ratio/high_mean": 0.013993188505992293, "clip_ratio/high_max": 0.013993188505992293, "clip_ratio/region_mean": 0.018400839995592833, "reward_total_mean": 0.998982846736908, "reward_meter_mean": 0.998982846736908, "reward_meter_std": 0.00020013147150166333, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.998982846736908, "reward_total_composite_std": 0.00020013147150166333} {"timestamp_utc": "2026-04-12T02:55:01Z", "mode": "train", "global_step": 2884, "epoch": 0.11583724946780737, "loss": 0.0006, "grad_norm": 2.683151960372925, "learning_rate": 1.2636363636363637e-06, "num_tokens": 6544901.0, "completions/mean_length": 67.625, "completions/min_length": 67.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.625, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9992280006408691, "rewards/meter/std": 0.00013651978224515915, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992280006408691, "rewards/total_composite/std": 0.00013651978224515915, "reward": 0.9992280006408691, "reward_std": 0.00013652884808834642, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019809167832136154, "sampling/sampling_logp_difference/max": 1.097256064414978, "sampling/importance_sampling_ratio/min": 0.3337857127189636, "sampling/importance_sampling_ratio/mean": 0.999349057674408, "sampling/importance_sampling_ratio/max": 1.5717170238494873, "entropy": 0.14564206078648567, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.014788191299885511, "clip_ratio/high_max": 0.014788191299885511, "clip_ratio/region_mean": 0.014788191299885511, "reward_total_mean": 0.9992280006408691, "reward_meter_mean": 0.9992280006408691, "reward_meter_std": 0.00013651978224515915, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992280006408691, "reward_total_composite_std": 0.00013651978224515915} {"timestamp_utc": "2026-04-12T02:55:06Z", "mode": "train", "global_step": 2885, "epoch": 0.11587741494959232, "loss": 0.0008, "grad_norm": 0.26706913113594055, "learning_rate": 1.2606060606060608e-06, "num_tokens": 6546912.0, "completions/mean_length": 71.375, "completions/min_length": 71.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.375, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9994363784790039, "rewards/meter/std": 2.7232801585341804e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994363784790039, "rewards/total_composite/std": 2.7232801585341804e-05, "reward": 0.9994363784790039, "reward_std": 2.7221845812164247e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010549667291343212, "sampling/sampling_logp_difference/max": 0.676356315612793, "sampling/importance_sampling_ratio/min": 0.5084663033485413, "sampling/importance_sampling_ratio/mean": 1.002943754196167, "sampling/importance_sampling_ratio/max": 1.2115890979766846, "entropy": 0.08360666781663895, "clip_ratio/low_mean": 0.0017361111240461469, "clip_ratio/low_min": 0.0017361111240461469, "clip_ratio/high_mean": 0.0034966744715347886, "clip_ratio/high_max": 0.0034966744715347886, "clip_ratio/region_mean": 0.0052327855955809355, "reward_total_mean": 0.9994363784790039, "reward_meter_mean": 0.9994363784790039, "reward_meter_std": 2.7232801585341804e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994363784790039, "reward_total_composite_std": 2.7232801585341804e-05} {"timestamp_utc": "2026-04-12T02:55:10Z", "mode": "train", "global_step": 2886, "epoch": 0.11591758043137727, "loss": 0.0053, "grad_norm": 2.8855912685394287, "learning_rate": 1.2575757575757578e-06, "num_tokens": 6548792.0, "completions/mean_length": 67.0, "completions/min_length": 66.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9954104423522949, "rewards/meter/std": 0.007760278414934874, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9954104423522949, "rewards/total_composite/std": 0.007760278414934874, "reward": 0.9954104423522949, "reward_std": 0.007760272361338139, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009755668230354786, "sampling/sampling_logp_difference/max": 0.9411745071411133, "sampling/importance_sampling_ratio/min": 0.39016932249069214, "sampling/importance_sampling_ratio/mean": 1.0019506216049194, "sampling/importance_sampling_ratio/max": 1.3913602828979492, "entropy": 0.09443793492391706, "clip_ratio/low_mean": 0.0036764706019312143, "clip_ratio/low_min": 0.0036764706019312143, "clip_ratio/high_mean": 0.0018656715983524919, "clip_ratio/high_max": 0.0018656715983524919, "clip_ratio/region_mean": 0.005542142200283706, "reward_total_mean": 0.9954104423522949, "reward_meter_mean": 0.9954104423522949, "reward_meter_std": 0.007760278414934874, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9954104423522949, "reward_total_composite_std": 0.007760278414934874} {"timestamp_utc": "2026-04-12T02:55:15Z", "mode": "train", "global_step": 2887, "epoch": 0.11595774591316223, "loss": -0.0051, "grad_norm": 3.4663033485412598, "learning_rate": 1.2545454545454546e-06, "num_tokens": 6550591.0, "completions/mean_length": 65.875, "completions/min_length": 65.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.875, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9924707412719727, "rewards/meter/std": 0.0012687068665400147, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9924707412719727, "rewards/total_composite/std": 0.0012687068665400147, "reward": 0.9924707412719727, "reward_std": 0.001268712687306106, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02419329807162285, "sampling/sampling_logp_difference/max": 0.6981816291809082, "sampling/importance_sampling_ratio/min": 0.4974890947341919, "sampling/importance_sampling_ratio/mean": 1.0042961835861206, "sampling/importance_sampling_ratio/max": 1.3884730339050293, "entropy": 0.17512462474405766, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/high_mean": 0.007520923274569213, "clip_ratio/high_max": 0.007520923274569213, "clip_ratio/region_mean": 0.0094148627249524, "reward_total_mean": 0.9924707412719727, "reward_meter_mean": 0.9924707412719727, "reward_meter_std": 0.0012687068665400147, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9924707412719727, "reward_total_composite_std": 0.0012687068665400147} {"timestamp_utc": "2026-04-12T02:55:20Z", "mode": "train", "global_step": 2888, "epoch": 0.11599791139494718, "loss": -0.001, "grad_norm": 0.5309938192367554, "learning_rate": 1.2515151515151517e-06, "num_tokens": 6552707.0, "completions/mean_length": 97.5, "completions/min_length": 96.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.5, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.9980380535125732, "rewards/meter/std": 4.4257390982238576e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980380535125732, "rewards/total_composite/std": 4.4257390982238576e-05, "reward": 0.9980380535125732, "reward_std": 4.425693259690888e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015023077838122845, "sampling/sampling_logp_difference/max": 0.9142270088195801, "sampling/importance_sampling_ratio/min": 0.4008263647556305, "sampling/importance_sampling_ratio/mean": 1.001752257347107, "sampling/importance_sampling_ratio/max": 1.4897899627685547, "entropy": 0.1272029634565115, "clip_ratio/low_mean": 0.003839680110104382, "clip_ratio/low_min": 0.003839680110104382, "clip_ratio/high_mean": 0.011493272730149329, "clip_ratio/high_max": 0.011493272730149329, "clip_ratio/region_mean": 0.01533295284025371, "reward_total_mean": 0.9980380535125732, "reward_meter_mean": 0.9980380535125732, "reward_meter_std": 4.4257390982238576e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9980380535125732, "reward_total_composite_std": 4.4257390982238576e-05} {"timestamp_utc": "2026-04-12T02:55:25Z", "mode": "train", "global_step": 2889, "epoch": 0.11603807687673214, "loss": -0.0002, "grad_norm": 1.1036797761917114, "learning_rate": 1.2484848484848485e-06, "num_tokens": 6554583.0, "completions/mean_length": 79.5, "completions/min_length": 78.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.5, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.998902440071106, "rewards/meter/std": 0.00011711295519489795, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998902440071106, "rewards/total_composite/std": 0.00011711295519489795, "reward": 0.998902440071106, "reward_std": 0.00011710206308634952, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02463226206600666, "sampling/sampling_logp_difference/max": 1.763596534729004, "sampling/importance_sampling_ratio/min": 0.17142720520496368, "sampling/importance_sampling_ratio/mean": 1.008751392364502, "sampling/importance_sampling_ratio/max": 1.4718900918960571, "entropy": 0.21664478071033955, "clip_ratio/low_mean": 0.004707766929641366, "clip_ratio/low_min": 0.004707766929641366, "clip_ratio/high_mean": 0.007871835259720683, "clip_ratio/high_max": 0.007871835259720683, "clip_ratio/region_mean": 0.012579602189362049, "reward_total_mean": 0.998902440071106, "reward_meter_mean": 0.998902440071106, "reward_meter_std": 0.00011711295519489795, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.998902440071106, "reward_total_composite_std": 0.00011711295519489795} {"timestamp_utc": "2026-04-12T02:55:30Z", "mode": "train", "global_step": 2890, "epoch": 0.11607824235851709, "loss": -0.0123, "grad_norm": 2.1610708236694336, "learning_rate": 1.2454545454545456e-06, "num_tokens": 6556527.0, "completions/mean_length": 92.0, "completions/min_length": 87.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.0, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.9974758625030518, "rewards/meter/std": 0.0004072287702001631, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9974758625030518, "rewards/total_composite/std": 0.0004072287702001631, "reward": 0.9974758625030518, "reward_std": 0.0004072203009855002, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01821918971836567, "sampling/sampling_logp_difference/max": 1.333749771118164, "sampling/importance_sampling_ratio/min": 0.26348739862442017, "sampling/importance_sampling_ratio/mean": 1.0036619901657104, "sampling/importance_sampling_ratio/max": 1.664230227470398, "entropy": 0.14203907828778028, "clip_ratio/low_mean": 0.005469039548188448, "clip_ratio/low_min": 0.005469039548188448, "clip_ratio/high_mean": 0.013470079400576651, "clip_ratio/high_max": 0.013470079400576651, "clip_ratio/region_mean": 0.0189391189487651, "reward_total_mean": 0.9974758625030518, "reward_meter_mean": 0.9974758625030518, "reward_meter_std": 0.0004072287702001631, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9974758625030518, "reward_total_composite_std": 0.0004072287702001631} {"timestamp_utc": "2026-04-12T02:55:35Z", "mode": "train", "global_step": 2891, "epoch": 0.11611840784030204, "loss": 0.0102, "grad_norm": 3.022761106491089, "learning_rate": 1.2424242424242424e-06, "num_tokens": 6558597.0, "completions/mean_length": 97.75, "completions/min_length": 96.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.75, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.9901492595672607, "rewards/meter/std": 0.005893507041037083, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9901492595672607, "rewards/total_composite/std": 0.005893507041037083, "reward": 0.9901492595672607, "reward_std": 0.005893507041037083, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.032126907259225845, "sampling/sampling_logp_difference/max": 1.2647018432617188, "sampling/importance_sampling_ratio/min": 0.2823234498500824, "sampling/importance_sampling_ratio/mean": 1.0117164850234985, "sampling/importance_sampling_ratio/max": 1.6783695220947266, "entropy": 0.2828730084002018, "clip_ratio/low_mean": 0.008739139186218381, "clip_ratio/low_min": 0.008739139186218381, "clip_ratio/high_mean": 0.01412348123267293, "clip_ratio/high_max": 0.01412348123267293, "clip_ratio/region_mean": 0.02286262041889131, "reward_total_mean": 0.9901492595672607, "reward_meter_mean": 0.9901492595672607, "reward_meter_std": 0.005893507041037083, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9901492595672607, "reward_total_composite_std": 0.005893507041037083} {"timestamp_utc": "2026-04-12T02:55:39Z", "mode": "train", "global_step": 2892, "epoch": 0.116158573322087, "loss": 0.0466, "grad_norm": 15.570990562438965, "learning_rate": 1.2393939393939394e-06, "num_tokens": 6559937.0, "completions/mean_length": 23.5, "completions/min_length": 21.0, "completions/max_length": 25.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.5, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.895869255065918, "rewards/meter/std": 0.033730946481227875, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.895869255065918, "rewards/total_composite/std": 0.033730946481227875, "reward": 0.895869255065918, "reward_std": 0.03373093158006668, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0897468775510788, "sampling/sampling_logp_difference/max": 1.4235849380493164, "sampling/importance_sampling_ratio/min": 0.24084904789924622, "sampling/importance_sampling_ratio/mean": 0.9834393858909607, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37175317015498877, "clip_ratio/low_mean": 0.020869565196335316, "clip_ratio/low_min": 0.020869565196335316, "clip_ratio/high_mean": 0.06117365509271622, "clip_ratio/high_max": 0.06117365509271622, "clip_ratio/region_mean": 0.08204322028905153, "reward_total_mean": 0.895869255065918, "reward_meter_mean": 0.895869255065918, "reward_meter_std": 0.033730946481227875, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.895869255065918, "reward_total_composite_std": 0.033730946481227875} {"timestamp_utc": "2026-04-12T02:55:43Z", "mode": "train", "global_step": 2893, "epoch": 0.11619873880387195, "loss": -0.0319, "grad_norm": 11.483199119567871, "learning_rate": 1.2363636363636365e-06, "num_tokens": 6561687.0, "completions/mean_length": 59.75, "completions/min_length": 56.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.75, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9967870712280273, "rewards/meter/std": 0.0010188753949478269, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9967870712280273, "rewards/total_composite/std": 0.0010188753949478269, "reward": 0.9967870712280273, "reward_std": 0.0010188753949478269, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01920030452311039, "sampling/sampling_logp_difference/max": 1.7169630527496338, "sampling/importance_sampling_ratio/min": 0.17961078882217407, "sampling/importance_sampling_ratio/mean": 1.0000895261764526, "sampling/importance_sampling_ratio/max": 1.8504060506820679, "entropy": 0.07222167262807488, "clip_ratio/low_mean": 0.006696428870782256, "clip_ratio/low_min": 0.006696428870782256, "clip_ratio/high_mean": 0.010245901066809893, "clip_ratio/high_max": 0.010245901066809893, "clip_ratio/region_mean": 0.01694232993759215, "reward_total_mean": 0.9967870712280273, "reward_meter_mean": 0.9967870712280273, "reward_meter_std": 0.0010188753949478269, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9967870712280273, "reward_total_composite_std": 0.0010188753949478269} {"timestamp_utc": "2026-04-12T02:55:49Z", "mode": "train", "global_step": 2894, "epoch": 0.1162389042856569, "loss": -0.0994, "grad_norm": 1.820000171661377, "learning_rate": 1.2333333333333335e-06, "num_tokens": 6564523.0, "completions/mean_length": 154.5, "completions/min_length": 108.0, "completions/max_length": 166.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 154.5, "completions/min_terminated_length": 108.0, "completions/max_terminated_length": 166.0, "rewards/meter/mean": 0.9889054298400879, "rewards/meter/std": 0.003125808434560895, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.1414213478565216, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.9117738604545593, "rewards/total_composite/std": 0.13682788610458374, "reward": 0.9117738604545593, "reward_std": 0.13682788610458374, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04758564755320549, "sampling/sampling_logp_difference/max": 1.404418706893921, "sampling/importance_sampling_ratio/min": 0.24550971388816833, "sampling/importance_sampling_ratio/mean": 1.008368730545044, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5260906405746937, "clip_ratio/low_mean": 0.007427186123095453, "clip_ratio/low_min": 0.007427186123095453, "clip_ratio/high_mean": 0.025422006146982312, "clip_ratio/high_max": 0.025422006146982312, "clip_ratio/region_mean": 0.032849192270077765, "reward_total_mean": 0.9117738604545593, "reward_meter_mean": 0.9889054298400879, "reward_meter_std": 0.003125808434560895, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.1414213478565216, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_total_composite_mean": 0.9117738604545593, "reward_total_composite_std": 0.13682788610458374} {"timestamp_utc": "2026-04-12T02:55:54Z", "mode": "train", "global_step": 2895, "epoch": 0.11627906976744186, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.2303030303030304e-06, "num_tokens": 6567027.0, "completions/mean_length": 132.0, "completions/min_length": 132.0, "completions/max_length": 132.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.0, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.6684597730636597, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7777777910232544, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5199131369590759, "rewards/total_composite/std": 0.0, "reward": 0.5199131369590759, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0006711529567837715, "sampling/sampling_logp_difference/max": 0.02449747547507286, "sampling/importance_sampling_ratio/min": 0.9979704022407532, "sampling/importance_sampling_ratio/mean": 1.0006672143936157, "sampling/importance_sampling_ratio/max": 1.0248000621795654, "entropy": 0.005916567693930119, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.5199131369590759, "reward_meter_mean": 0.6684597730636597, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7777777910232544, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5199131369590759, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:55:58Z", "mode": "train", "global_step": 2896, "epoch": 0.11631923524922681, "loss": 0.0006, "grad_norm": 3.799595355987549, "learning_rate": 1.2272727272727274e-06, "num_tokens": 6568899.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9972846508026123, "rewards/meter/std": 0.00015232501027639955, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9972846508026123, "rewards/total_composite/std": 0.00015232501027639955, "reward": 0.9972846508026123, "reward_std": 0.00015233403246384114, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005107899196445942, "sampling/sampling_logp_difference/max": 0.5471892356872559, "sampling/importance_sampling_ratio/min": 0.5785737633705139, "sampling/importance_sampling_ratio/mean": 1.0012083053588867, "sampling/importance_sampling_ratio/max": 1.18209969997406, "entropy": 0.035036823246628046, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.004098360426723957, "clip_ratio/high_max": 0.004098360426723957, "clip_ratio/region_mean": 0.004098360426723957, "reward_total_mean": 0.9972846508026123, "reward_meter_mean": 0.9972846508026123, "reward_meter_std": 0.00015232501027639955, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9972846508026123, "reward_total_composite_std": 0.00015232501027639955} {"timestamp_utc": "2026-04-12T02:56:03Z", "mode": "train", "global_step": 2897, "epoch": 0.11635940073101177, "loss": 0.0028, "grad_norm": 1.706589937210083, "learning_rate": 1.2242424242424242e-06, "num_tokens": 6571375.0, "completions/mean_length": 128.5, "completions/min_length": 127.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.5, "completions/min_terminated_length": 127.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9972264766693115, "rewards/meter/std": 0.002113205147907138, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9285714626312256, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.9259439706802368, "rewards/total_composite/std": 0.07545064389705658, "reward": 0.9259439706802368, "reward_std": 0.07545064389705658, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016268158331513405, "sampling/sampling_logp_difference/max": 0.9007883071899414, "sampling/importance_sampling_ratio/min": 0.4062493145465851, "sampling/importance_sampling_ratio/mean": 1.0051052570343018, "sampling/importance_sampling_ratio/max": 1.4906891584396362, "entropy": 0.17375356331467628, "clip_ratio/low_mean": 0.005829093977808952, "clip_ratio/low_min": 0.005829093977808952, "clip_ratio/high_mean": 0.008736189920455217, "clip_ratio/high_max": 0.008736189920455217, "clip_ratio/region_mean": 0.01456528389826417, "reward_total_mean": 0.9259439706802368, "reward_meter_mean": 0.9972264766693115, "reward_meter_std": 0.002113205147907138, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9285714626312256, "reward_repeat_penalty_std": 0.07636035233736038, "reward_total_composite_mean": 0.9259439706802368, "reward_total_composite_std": 0.07545064389705658} {"timestamp_utc": "2026-04-12T02:56:07Z", "mode": "train", "global_step": 2898, "epoch": 0.11639956621279672, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.2212121212121213e-06, "num_tokens": 6573071.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002584453613962978, "sampling/sampling_logp_difference/max": 0.004060388542711735, "sampling/importance_sampling_ratio/min": 0.9999027252197266, "sampling/importance_sampling_ratio/mean": 1.0002572536468506, "sampling/importance_sampling_ratio/max": 1.0040686130523682, "entropy": 0.0020958341483492404, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:56:12Z", "mode": "train", "global_step": 2899, "epoch": 0.11643973169458167, "loss": -0.0001, "grad_norm": 0.039646849036216736, "learning_rate": 1.2181818181818183e-06, "num_tokens": 6574871.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9973360300064087, "rewards/meter/std": 8.113272997434251e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973360300064087, "rewards/total_composite/std": 8.113272997434251e-06, "reward": 0.9973360300064087, "reward_std": 8.10424353403505e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004976366646587849, "sampling/sampling_logp_difference/max": 0.4815742075443268, "sampling/importance_sampling_ratio/min": 0.617810070514679, "sampling/importance_sampling_ratio/mean": 1.0008740425109863, "sampling/importance_sampling_ratio/max": 1.1416552066802979, "entropy": 0.034302944084629416, "clip_ratio/low_mean": 0.004098360426723957, "clip_ratio/low_min": 0.004098360426723957, "clip_ratio/high_mean": 0.004098360426723957, "clip_ratio/high_max": 0.004098360426723957, "clip_ratio/region_mean": 0.008196720853447914, "reward_total_mean": 0.9973360300064087, "reward_meter_mean": 0.9973360300064087, "reward_meter_std": 8.113272997434251e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9973360300064087, "reward_total_composite_std": 8.113272997434251e-06} {"timestamp_utc": "2026-04-12T02:56:16Z", "mode": "train", "global_step": 2900, "epoch": 0.11647989717636663, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.2151515151515154e-06, "num_tokens": 6576751.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002209599915659055, "sampling/sampling_logp_difference/max": 0.009038996882736683, "sampling/importance_sampling_ratio/min": 0.9910017251968384, "sampling/importance_sampling_ratio/mean": 1.000099539756775, "sampling/importance_sampling_ratio/max": 1.006137490272522, "entropy": 0.0027679620689013973, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:57:35Z", "mode": "eval", "global_step": 2900, "epoch": 0.11647989717636663, "eval_loss": NaN, "eval_runtime": 78.4008, "eval_samples_per_second": 1.327, "eval_steps_per_second": 0.166, "eval_num_tokens": 6576751.0, "eval_completions/mean_length": 211.68269230769232, "eval_completions/min_length": 61.84615384615385, "eval_completions/max_length": 422.84615384615387, "eval_completions/clipped_ratio": 0.019230769230769232, "eval_completions/mean_terminated_length": 205.16483600323016, "eval_completions/min_terminated_length": 61.84615384615385, "eval_completions/max_terminated_length": 402.46153846153845, "eval_rewards/meter/mean": 0.7883604077192453, "eval_rewards/meter/std": 0.34034269847548926, "eval_rewards/count_adherence/mean": 0.9613341138913081, "eval_rewards/count_adherence/std": 0.06466435583738181, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/repeat_penalty/mean": 0.955476293197045, "eval_rewards/repeat_penalty/std": 0.07249205201291122, "eval_rewards/total_composite/mean": 0.7234287353662344, "eval_rewards/total_composite/std": 0.343703310077007, "eval_reward": 0.7234287353662344, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.03452126896725251, "eval_sampling/sampling_logp_difference/max": 1.1897428219134991, "eval_sampling/importance_sampling_ratio/min": 0.3125262191662422, "eval_sampling/importance_sampling_ratio/mean": 1.0095937985640306, "eval_sampling/importance_sampling_ratio/max": 1.529663553604713, "eval_entropy": 0.4129388790864211, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7234287353662344, "eval_reward_meter_mean": 0.7883604077192453, "eval_reward_meter_std": 0.34034269847548926, "eval_reward_count_adherence_mean": 0.9613341138913081, "eval_reward_count_adherence_std": 0.06466435583738181, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_repeat_penalty_mean": 0.955476293197045, "eval_reward_repeat_penalty_std": 0.07249205201291122, "eval_reward_total_composite_mean": 0.7234287353662344, "eval_reward_total_composite_std": 0.343703310077007} {"timestamp_utc": "2026-04-12T02:57:42Z", "mode": "train", "global_step": 2901, "epoch": 0.11652006265815158, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.2121212121212122e-06, "num_tokens": 6578415.0, "completions/mean_length": 35.0, "completions/min_length": 35.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9980231523513794, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980231523513794, "rewards/total_composite/std": 0.0, "reward": 0.9980231523513794, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0006013626698404551, "sampling/sampling_logp_difference/max": 0.007742481306195259, "sampling/importance_sampling_ratio/min": 0.9922874569892883, "sampling/importance_sampling_ratio/mean": 1.0005377531051636, "sampling/importance_sampling_ratio/max": 1.0070579051971436, "entropy": 0.005171062133740634, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9980231523513794, "reward_meter_mean": 0.9980231523513794, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9980231523513794, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:57:53Z", "mode": "train", "global_step": 2902, "epoch": 0.11656022813993654, "loss": 0.2202, "grad_norm": 13.265780448913574, "learning_rate": 1.2090909090909092e-06, "num_tokens": 6582881.0, "completions/mean_length": 501.25, "completions/min_length": 485.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 497.66668701171875, "completions/min_terminated_length": 485.0, "completions/max_terminated_length": 512.0, "rewards/meter/mean": 0.9986035823822021, "rewards/meter/std": 0.0011026532156392932, "rewards/count_adherence/mean": 0.796875, "rewards/count_adherence/std": 0.04419417306780815, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9701460003852844, "rewards/repeat_penalty/std": 0.01849004067480564, "rewards/total_composite/mean": 0.7718735337257385, "rewards/total_composite/std": 0.04241907224059105, "reward": 0.7718735337257385, "reward_std": 0.042419061064720154, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04629309102892876, "sampling/sampling_logp_difference/max": 2.552377700805664, "sampling/importance_sampling_ratio/min": 0.07789622992277145, "sampling/importance_sampling_ratio/mean": 1.0130130052566528, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3398243226110935, "clip_ratio/low_mean": 0.010757095878943801, "clip_ratio/low_min": 0.010757095878943801, "clip_ratio/high_mean": 0.010358819272369146, "clip_ratio/high_max": 0.010358819272369146, "clip_ratio/region_mean": 0.021115915151312947, "reward_total_mean": 0.7718735337257385, "reward_meter_mean": 0.9986035823822021, "reward_meter_std": 0.0011026532156392932, "reward_count_adherence_mean": 0.796875, "reward_count_adherence_std": 0.04419417306780815, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9701460003852844, "reward_repeat_penalty_std": 0.01849004067480564, "reward_total_composite_mean": 0.7718735337257385, "reward_total_composite_std": 0.04241907224059105} {"timestamp_utc": "2026-04-12T02:58:01Z", "mode": "train", "global_step": 2903, "epoch": 0.11660039362172149, "loss": -0.0437, "grad_norm": 2.858194351196289, "learning_rate": 1.206060606060606e-06, "num_tokens": 6587110.0, "completions/mean_length": 325.625, "completions/min_length": 304.0, "completions/max_length": 349.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 325.625, "completions/min_terminated_length": 304.0, "completions/max_terminated_length": 349.0, "rewards/meter/mean": 0.9878900051116943, "rewards/meter/std": 0.025390086695551872, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.0534522607922554, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9380475282669067, "rewards/total_composite/std": 0.04961733520030975, "reward": 0.9380475282669067, "reward_std": 0.049617357552051544, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06375819444656372, "sampling/sampling_logp_difference/max": 2.556863784790039, "sampling/importance_sampling_ratio/min": 0.0775475725531578, "sampling/importance_sampling_ratio/mean": 1.017285943031311, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.633208267390728, "clip_ratio/low_mean": 0.025543570518493652, "clip_ratio/low_min": 0.025543570518493652, "clip_ratio/high_mean": 0.01695435866713524, "clip_ratio/high_max": 0.01695435866713524, "clip_ratio/region_mean": 0.04249792918562889, "reward_total_mean": 0.9380475282669067, "reward_meter_mean": 0.9878900051116943, "reward_meter_std": 0.025390086695551872, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.0534522607922554, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9380475282669067, "reward_total_composite_std": 0.04961733520030975} {"timestamp_utc": "2026-04-12T02:58:05Z", "mode": "train", "global_step": 2904, "epoch": 0.11664055910350644, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.2030303030303031e-06, "num_tokens": 6588910.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00021447156905196607, "sampling/sampling_logp_difference/max": 0.01532711274921894, "sampling/importance_sampling_ratio/min": 0.9847897291183472, "sampling/importance_sampling_ratio/mean": 1.0000479221343994, "sampling/importance_sampling_ratio/max": 1.0044045448303223, "entropy": 0.0017551528289914131, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:58:10Z", "mode": "train", "global_step": 2905, "epoch": 0.1166807245852914, "loss": 0.0003, "grad_norm": 0.42203962802886963, "learning_rate": 1.2000000000000002e-06, "num_tokens": 6591138.0, "completions/mean_length": 97.5, "completions/min_length": 96.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.5, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9980453252792358, "rewards/meter/std": 4.257191903889179e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980453252792358, "rewards/total_composite/std": 4.257191903889179e-05, "reward": 0.9980453252792358, "reward_std": 4.25745893153362e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014815653674304485, "sampling/sampling_logp_difference/max": 1.1947669982910156, "sampling/importance_sampling_ratio/min": 0.30277448892593384, "sampling/importance_sampling_ratio/mean": 1.002492904663086, "sampling/importance_sampling_ratio/max": 1.5006216764450073, "entropy": 0.11437637638300657, "clip_ratio/low_mean": 0.005181760177947581, "clip_ratio/low_min": 0.005181760177947581, "clip_ratio/high_mean": 0.008968020090833306, "clip_ratio/high_max": 0.008968020090833306, "clip_ratio/region_mean": 0.014149780268780887, "reward_total_mean": 0.9980453252792358, "reward_meter_mean": 0.9980453252792358, "reward_meter_std": 4.257191903889179e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9980453252792358, "reward_total_composite_std": 4.257191903889179e-05} {"timestamp_utc": "2026-04-12T02:58:15Z", "mode": "train", "global_step": 2906, "epoch": 0.11672089006707635, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.196969696969697e-06, "num_tokens": 6592674.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00027867371682077646, "sampling/sampling_logp_difference/max": 0.003768636379390955, "sampling/importance_sampling_ratio/min": 0.9966413378715515, "sampling/importance_sampling_ratio/mean": 1.0002539157867432, "sampling/importance_sampling_ratio/max": 1.003775715827942, "entropy": 0.002388683380559087, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:58:19Z", "mode": "train", "global_step": 2907, "epoch": 0.1167610555488613, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.193939393939394e-06, "num_tokens": 6594410.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00022611531312577426, "sampling/sampling_logp_difference/max": 0.0029838387854397297, "sampling/importance_sampling_ratio/min": 0.9999564290046692, "sampling/importance_sampling_ratio/mean": 1.0002257823944092, "sampling/importance_sampling_ratio/max": 1.002988338470459, "entropy": 0.001633810141356662, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:58:23Z", "mode": "train", "global_step": 2908, "epoch": 0.11680122103064626, "loss": -0.0005, "grad_norm": 0.9043018221855164, "learning_rate": 1.190909090909091e-06, "num_tokens": 6595818.0, "completions/mean_length": 35.0, "completions/min_length": 35.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9995806217193604, "rewards/meter/std": 1.9758595954044722e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9995806217193604, "rewards/total_composite/std": 1.9758595954044722e-05, "reward": 0.9995806217193604, "reward_std": 1.9746988982660696e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005839377176016569, "sampling/sampling_logp_difference/max": 0.7329778671264648, "sampling/importance_sampling_ratio/min": 0.4804760813713074, "sampling/importance_sampling_ratio/mean": 1.0006860494613647, "sampling/importance_sampling_ratio/max": 1.241074562072754, "entropy": 0.03689346136525273, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9995806217193604, "reward_meter_mean": 0.9995806217193604, "reward_meter_std": 1.9758595954044722e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9995806217193604, "reward_total_composite_std": 1.9758595954044722e-05} {"timestamp_utc": "2026-04-12T02:58:31Z", "mode": "train", "global_step": 2909, "epoch": 0.11684138651243121, "loss": -0.0212, "grad_norm": 1.9418765306472778, "learning_rate": 1.187878787878788e-06, "num_tokens": 6600002.0, "completions/mean_length": 306.0, "completions/min_length": 284.0, "completions/max_length": 319.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 306.0, "completions/min_terminated_length": 284.0, "completions/max_terminated_length": 319.0, "rewards/meter/mean": 0.973264217376709, "rewards/meter/std": 0.044348716735839844, "rewards/count_adherence/mean": 0.887499988079071, "rewards/count_adherence/std": 0.035355325788259506, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.962775707244873, "rewards/repeat_penalty/std": 0.044049203395843506, "rewards/total_composite/mean": 0.7390196323394775, "rewards/total_composite/std": 0.3020605444908142, "reward": 0.7390196323394775, "reward_std": 0.3020605444908142, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.048917196691036224, "sampling/sampling_logp_difference/max": 1.7572402954101562, "sampling/importance_sampling_ratio/min": 0.17252032458782196, "sampling/importance_sampling_ratio/mean": 1.0099446773529053, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5155597105622292, "clip_ratio/low_mean": 0.0035211266949772835, "clip_ratio/low_min": 0.0035211266949772835, "clip_ratio/high_mean": 0.032285032561048865, "clip_ratio/high_max": 0.032285032561048865, "clip_ratio/region_mean": 0.03580615925602615, "reward_total_mean": 0.7390196323394775, "reward_meter_mean": 0.973264217376709, "reward_meter_std": 0.044348716735839844, "reward_count_adherence_mean": 0.887499988079071, "reward_count_adherence_std": 0.035355325788259506, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.962775707244873, "reward_repeat_penalty_std": 0.044049203395843506, "reward_total_composite_mean": 0.7390196323394775, "reward_total_composite_std": 0.3020605444908142} {"timestamp_utc": "2026-04-12T02:58:35Z", "mode": "train", "global_step": 2910, "epoch": 0.11688155199421617, "loss": 0.0001, "grad_norm": 0.2488185316324234, "learning_rate": 1.184848484848485e-06, "num_tokens": 6601362.0, "completions/mean_length": 35.0, "completions/min_length": 35.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9995900392532349, "rewards/meter/std": 2.9798916330037173e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9995900392532349, "rewards/total_composite/std": 2.9798916330037173e-06, "reward": 0.9995900392532349, "reward_std": 2.9798916330037173e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0049346331506967545, "sampling/sampling_logp_difference/max": 0.7781157493591309, "sampling/importance_sampling_ratio/min": 0.4592705965042114, "sampling/importance_sampling_ratio/mean": 0.9999272227287292, "sampling/importance_sampling_ratio/max": 1.1393342018127441, "entropy": 0.02543548052199185, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9995900392532349, "reward_meter_mean": 0.9995900392532349, "reward_meter_std": 2.9798916330037173e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9995900392532349, "reward_total_composite_std": 2.9798916330037173e-06} {"timestamp_utc": "2026-04-12T02:58:39Z", "mode": "train", "global_step": 2911, "epoch": 0.11692171747600112, "loss": -0.0001, "grad_norm": 0.16482257843017578, "learning_rate": 1.181818181818182e-06, "num_tokens": 6603186.0, "completions/mean_length": 67.0, "completions/min_length": 67.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981510639190674, "rewards/meter/std": 1.0189157364948187e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981510639190674, "rewards/total_composite/std": 1.0189157364948187e-05, "reward": 0.9981510639190674, "reward_std": 1.0199873031524476e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005824689287692308, "sampling/sampling_logp_difference/max": 0.6331427097320557, "sampling/importance_sampling_ratio/min": 0.530920684337616, "sampling/importance_sampling_ratio/mean": 0.9997050762176514, "sampling/importance_sampling_ratio/max": 1.1907520294189453, "entropy": 0.03822743846103549, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9981510639190674, "reward_meter_mean": 0.9981510639190674, "reward_meter_std": 1.0189157364948187e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981510639190674, "reward_total_composite_std": 1.0189157364948187e-05} {"timestamp_utc": "2026-04-12T02:58:43Z", "mode": "train", "global_step": 2912, "epoch": 0.11696188295778608, "loss": -0.0024, "grad_norm": 1.688647747039795, "learning_rate": 1.1787878787878788e-06, "num_tokens": 6605005.0, "completions/mean_length": 67.375, "completions/min_length": 66.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.375, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.99924635887146, "rewards/meter/std": 0.0001372320402879268, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.99924635887146, "rewards/total_composite/std": 0.0001372320402879268, "reward": 0.99924635887146, "reward_std": 0.0001372358383378014, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01559353992342949, "sampling/sampling_logp_difference/max": 1.1167762279510498, "sampling/importance_sampling_ratio/min": 0.32733336091041565, "sampling/importance_sampling_ratio/mean": 1.0067250728607178, "sampling/importance_sampling_ratio/max": 1.6964365243911743, "entropy": 0.12552092969417572, "clip_ratio/low_mean": 0.0074360816506668925, "clip_ratio/low_min": 0.0074360816506668925, "clip_ratio/high_mean": 0.0018656715983524919, "clip_ratio/high_max": 0.0018656715983524919, "clip_ratio/region_mean": 0.009301753249019384, "reward_total_mean": 0.99924635887146, "reward_meter_mean": 0.99924635887146, "reward_meter_std": 0.0001372320402879268, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.99924635887146, "reward_total_composite_std": 0.0001372320402879268} {"timestamp_utc": "2026-04-12T02:58:48Z", "mode": "train", "global_step": 2913, "epoch": 0.11700204843957103, "loss": -0.0167, "grad_norm": 5.435182094573975, "learning_rate": 1.1757575757575759e-06, "num_tokens": 6606638.0, "completions/mean_length": 69.125, "completions/min_length": 66.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.125, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9914849996566772, "rewards/meter/std": 0.013738686218857765, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9914849996566772, "rewards/total_composite/std": 0.013738686218857765, "reward": 0.9914849996566772, "reward_std": 0.013738670386373997, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027403326705098152, "sampling/sampling_logp_difference/max": 1.0196411609649658, "sampling/importance_sampling_ratio/min": 0.3607243597507477, "sampling/importance_sampling_ratio/mean": 1.0020140409469604, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23639377392828465, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.024883363395929337, "clip_ratio/high_max": 0.024883363395929337, "clip_ratio/region_mean": 0.024883363395929337, "reward_total_mean": 0.9914849996566772, "reward_meter_mean": 0.9914849996566772, "reward_meter_std": 0.013738686218857765, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9914849996566772, "reward_total_composite_std": 0.013738686218857765} {"timestamp_utc": "2026-04-12T02:58:55Z", "mode": "train", "global_step": 2914, "epoch": 0.11704221392135598, "loss": 0.004, "grad_norm": 2.8395214080810547, "learning_rate": 1.172727272727273e-06, "num_tokens": 6610549.0, "completions/mean_length": 302.875, "completions/min_length": 291.0, "completions/max_length": 312.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 302.875, "completions/min_terminated_length": 291.0, "completions/max_terminated_length": 312.0, "rewards/meter/mean": 0.9837737083435059, "rewards/meter/std": 0.04140230268239975, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.9926470518112183, "rewards/repeat_penalty/std": 0.020797256380319595, "rewards/total_composite/mean": 0.8662627935409546, "rewards/total_composite/std": 0.3506163954734802, "reward": 0.8662627935409546, "reward_std": 0.3506163954734802, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.059933777898550034, "sampling/sampling_logp_difference/max": 1.6421480178833008, "sampling/importance_sampling_ratio/min": 0.1935638189315796, "sampling/importance_sampling_ratio/mean": 1.010536551475525, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6028463244438171, "clip_ratio/low_mean": 0.004125412553548813, "clip_ratio/low_min": 0.004125412553548813, "clip_ratio/high_mean": 0.052266900427639484, "clip_ratio/high_max": 0.052266900427639484, "clip_ratio/region_mean": 0.0563923129811883, "reward_total_mean": 0.8662627935409546, "reward_meter_mean": 0.9837737083435059, "reward_meter_std": 0.04140230268239975, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.9926470518112183, "reward_repeat_penalty_std": 0.020797256380319595, "reward_total_composite_mean": 0.8662627935409546, "reward_total_composite_std": 0.3506163954734802} {"timestamp_utc": "2026-04-12T02:59:02Z", "mode": "train", "global_step": 2915, "epoch": 0.11708237940314094, "loss": 0.0239, "grad_norm": 2.1781363487243652, "learning_rate": 1.1696969696969697e-06, "num_tokens": 6614339.0, "completions/mean_length": 264.75, "completions/min_length": 256.0, "completions/max_length": 286.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 264.75, "completions/min_terminated_length": 256.0, "completions/max_terminated_length": 286.0, "rewards/meter/mean": 0.9970918893814087, "rewards/meter/std": 0.005339814815670252, "rewards/count_adherence/mean": 0.984375, "rewards/count_adherence/std": 0.04419417306780815, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9676470756530762, "rewards/repeat_penalty/std": 0.03468189761042595, "rewards/total_composite/mean": 0.9504234790802002, "rewards/total_composite/std": 0.06568735092878342, "reward": 0.9504234790802002, "reward_std": 0.06568736582994461, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.033314015716314316, "sampling/sampling_logp_difference/max": 1.4124369621276855, "sampling/importance_sampling_ratio/min": 0.24354904890060425, "sampling/importance_sampling_ratio/mean": 1.0066348314285278, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3080044835805893, "clip_ratio/low_mean": 0.017094367765821517, "clip_ratio/low_min": 0.017094367765821517, "clip_ratio/high_mean": 0.012379350140690804, "clip_ratio/high_max": 0.012379350140690804, "clip_ratio/region_mean": 0.02947371790651232, "reward_total_mean": 0.9504234790802002, "reward_meter_mean": 0.9970918893814087, "reward_meter_std": 0.005339814815670252, "reward_count_adherence_mean": 0.984375, "reward_count_adherence_std": 0.04419417306780815, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9676470756530762, "reward_repeat_penalty_std": 0.03468189761042595, "reward_total_composite_mean": 0.9504234790802002, "reward_total_composite_std": 0.06568735092878342} {"timestamp_utc": "2026-04-12T02:59:08Z", "mode": "train", "global_step": 2916, "epoch": 0.11712254488492589, "loss": -0.0014, "grad_norm": 1.5669199228286743, "learning_rate": 1.1666666666666668e-06, "num_tokens": 6617355.0, "completions/mean_length": 198.0, "completions/min_length": 196.0, "completions/max_length": 201.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 198.0, "completions/min_terminated_length": 196.0, "completions/max_terminated_length": 201.0, "rewards/meter/mean": 0.9989873170852661, "rewards/meter/std": 0.000616980018094182, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989873170852661, "rewards/total_composite/std": 0.000616980018094182, "reward": 0.9989873170852661, "reward_std": 0.0006169742555357516, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03752731904387474, "sampling/sampling_logp_difference/max": 1.0977716445922852, "sampling/importance_sampling_ratio/min": 0.33361366391181946, "sampling/importance_sampling_ratio/mean": 1.0102366209030151, "sampling/importance_sampling_ratio/max": 1.8285475969314575, "entropy": 0.3740917034447193, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/high_mean": 0.020882656681351364, "clip_ratio/high_max": 0.020882656681351364, "clip_ratio/region_mean": 0.02277659613173455, "reward_total_mean": 0.9989873170852661, "reward_meter_mean": 0.9989873170852661, "reward_meter_std": 0.000616980018094182, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9989873170852661, "reward_total_composite_std": 0.000616980018094182} {"timestamp_utc": "2026-04-12T02:59:13Z", "mode": "train", "global_step": 2917, "epoch": 0.11716271036671085, "loss": -0.0003, "grad_norm": 2.548100233078003, "learning_rate": 1.1636363636363638e-06, "num_tokens": 6619535.0, "completions/mean_length": 92.5, "completions/min_length": 91.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.5, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.9972745180130005, "rewards/meter/std": 0.00045341101940721273, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9972745180130005, "rewards/total_composite/std": 0.00045341101940721273, "reward": 0.9972745180130005, "reward_std": 0.00045341774239204824, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016107600182294846, "sampling/sampling_logp_difference/max": 1.0648202896118164, "sampling/importance_sampling_ratio/min": 0.3447898030281067, "sampling/importance_sampling_ratio/mean": 1.0014476776123047, "sampling/importance_sampling_ratio/max": 1.420615792274475, "entropy": 0.12023854069411755, "clip_ratio/low_mean": 0.008138206438161433, "clip_ratio/low_min": 0.008138206438161433, "clip_ratio/high_mean": 0.006749970489181578, "clip_ratio/high_max": 0.006749970489181578, "clip_ratio/region_mean": 0.014888176927343011, "reward_total_mean": 0.9972745180130005, "reward_meter_mean": 0.9972745180130005, "reward_meter_std": 0.00045341101940721273, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9972745180130005, "reward_total_composite_std": 0.00045341101940721273} {"timestamp_utc": "2026-04-12T02:59:18Z", "mode": "train", "global_step": 2918, "epoch": 0.1172028758484958, "loss": 0.0044, "grad_norm": 3.6516404151916504, "learning_rate": 1.1606060606060607e-06, "num_tokens": 6621518.0, "completions/mean_length": 103.875, "completions/min_length": 101.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.875, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.9887614250183105, "rewards/meter/std": 0.016851743683218956, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9887614250183105, "rewards/total_composite/std": 0.016851743683218956, "reward": 0.9887614250183105, "reward_std": 0.01685173623263836, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.038346562534570694, "sampling/sampling_logp_difference/max": 1.4495925903320312, "sampling/importance_sampling_ratio/min": 0.2346658855676651, "sampling/importance_sampling_ratio/mean": 1.0085445642471313, "sampling/importance_sampling_ratio/max": 1.9795823097229004, "entropy": 0.2945725992321968, "clip_ratio/low_mean": 0.0024271844886243343, "clip_ratio/low_min": 0.0024271844886243343, "clip_ratio/high_mean": 0.019217372057028115, "clip_ratio/high_max": 0.019217372057028115, "clip_ratio/region_mean": 0.02164455654565245, "reward_total_mean": 0.9887614250183105, "reward_meter_mean": 0.9887614250183105, "reward_meter_std": 0.016851743683218956, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9887614250183105, "reward_total_composite_std": 0.016851743683218956} {"timestamp_utc": "2026-04-12T02:59:23Z", "mode": "train", "global_step": 2919, "epoch": 0.11724304133028075, "loss": -0.0035, "grad_norm": 1.5092811584472656, "learning_rate": 1.1575757575757577e-06, "num_tokens": 6623884.0, "completions/mean_length": 106.75, "completions/min_length": 106.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.75, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.999142587184906, "rewards/meter/std": 0.0002281271299580112, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999142587184906, "rewards/total_composite/std": 0.0002281271299580112, "reward": 0.999142587184906, "reward_std": 0.00022812940005678684, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016602281481027603, "sampling/sampling_logp_difference/max": 1.0120248794555664, "sampling/importance_sampling_ratio/min": 0.3634822368621826, "sampling/importance_sampling_ratio/mean": 1.0026243925094604, "sampling/importance_sampling_ratio/max": 1.5648819208145142, "entropy": 0.14118101634085178, "clip_ratio/low_mean": 0.004705960163846612, "clip_ratio/low_min": 0.004705960163846612, "clip_ratio/high_mean": 0.003504672786220908, "clip_ratio/high_max": 0.003504672786220908, "clip_ratio/region_mean": 0.00821063295006752, "reward_total_mean": 0.999142587184906, "reward_meter_mean": 0.999142587184906, "reward_meter_std": 0.0002281271299580112, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.999142587184906, "reward_total_composite_std": 0.0002281271299580112} {"timestamp_utc": "2026-04-12T02:59:28Z", "mode": "train", "global_step": 2920, "epoch": 0.11728320681206571, "loss": 0.002, "grad_norm": 0.6831678152084351, "learning_rate": 1.1545454545454545e-06, "num_tokens": 6626176.0, "completions/mean_length": 106.5, "completions/min_length": 106.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.5, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.9992160201072693, "rewards/meter/std": 5.9427493397379294e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992160201072693, "rewards/total_composite/std": 5.9427493397379294e-05, "reward": 0.9992160201072693, "reward_std": 5.943301584920846e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013828075490891933, "sampling/sampling_logp_difference/max": 0.6171674728393555, "sampling/importance_sampling_ratio/min": 0.5394703149795532, "sampling/importance_sampling_ratio/mean": 1.0034005641937256, "sampling/importance_sampling_ratio/max": 1.5806970596313477, "entropy": 0.13282244093716145, "clip_ratio/low_mean": 0.002358490601181984, "clip_ratio/low_min": 0.002358490601181984, "clip_ratio/high_mean": 0.00352671486325562, "clip_ratio/high_max": 0.00352671486325562, "clip_ratio/region_mean": 0.005885205464437604, "reward_total_mean": 0.9992160201072693, "reward_meter_mean": 0.9992160201072693, "reward_meter_std": 5.9427493397379294e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992160201072693, "reward_total_composite_std": 5.9427493397379294e-05} {"timestamp_utc": "2026-04-12T02:59:33Z", "mode": "train", "global_step": 2921, "epoch": 0.11732337229385066, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.1515151515151516e-06, "num_tokens": 6627968.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0001586712896823883, "sampling/sampling_logp_difference/max": 0.006689060479402542, "sampling/importance_sampling_ratio/min": 0.9933332204818726, "sampling/importance_sampling_ratio/mean": 1.0000739097595215, "sampling/importance_sampling_ratio/max": 1.0063046216964722, "entropy": 0.0019735290406970307, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T02:59:37Z", "mode": "train", "global_step": 2922, "epoch": 0.11736353777563562, "loss": 0.0033, "grad_norm": 0.311759889125824, "learning_rate": 1.1484848484848486e-06, "num_tokens": 6629489.0, "completions/mean_length": 34.125, "completions/min_length": 34.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.125, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9968487024307251, "rewards/meter/std": 0.0025374931283295155, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9968487024307251, "rewards/total_composite/std": 0.0025374931283295155, "reward": 0.9968487024307251, "reward_std": 0.0025374931283295155, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006284201052039862, "sampling/sampling_logp_difference/max": 0.8642997741699219, "sampling/importance_sampling_ratio/min": 0.4213464856147766, "sampling/importance_sampling_ratio/mean": 1.0004445314407349, "sampling/importance_sampling_ratio/max": 1.059367060661316, "entropy": 0.03445210144855082, "clip_ratio/low_mean": 0.0035714285913854837, "clip_ratio/low_min": 0.0035714285913854837, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0035714285913854837, "reward_total_mean": 0.9968487024307251, "reward_meter_mean": 0.9968487024307251, "reward_meter_std": 0.0025374931283295155, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9968487024307251, "reward_total_composite_std": 0.0025374931283295155} {"timestamp_utc": "2026-04-12T02:59:43Z", "mode": "train", "global_step": 2923, "epoch": 0.11740370325742057, "loss": 0.0051, "grad_norm": 4.833311557769775, "learning_rate": 1.1454545454545457e-06, "num_tokens": 6632353.0, "completions/mean_length": 168.0, "completions/min_length": 164.0, "completions/max_length": 173.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 168.0, "completions/min_terminated_length": 164.0, "completions/max_terminated_length": 173.0, "rewards/meter/mean": 0.9983302354812622, "rewards/meter/std": 0.0015840368578210473, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9983302354812622, "rewards/total_composite/std": 0.0015840368578210473, "reward": 0.9983302354812622, "reward_std": 0.0015840278938412666, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04776981472969055, "sampling/sampling_logp_difference/max": 1.0252695083618164, "sampling/importance_sampling_ratio/min": 0.3586997985839844, "sampling/importance_sampling_ratio/mean": 1.0118147134780884, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4091906175017357, "clip_ratio/low_mean": 0.00808823574334383, "clip_ratio/low_min": 0.00808823574334383, "clip_ratio/high_mean": 0.03803631942719221, "clip_ratio/high_max": 0.03803631942719221, "clip_ratio/region_mean": 0.04612455517053604, "reward_total_mean": 0.9983302354812622, "reward_meter_mean": 0.9983302354812622, "reward_meter_std": 0.0015840368578210473, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9983302354812622, "reward_total_composite_std": 0.0015840368578210473} {"timestamp_utc": "2026-04-12T02:59:47Z", "mode": "train", "global_step": 2924, "epoch": 0.11744386873920552, "loss": -0.0, "grad_norm": 0.0032307966612279415, "learning_rate": 1.1424242424242425e-06, "num_tokens": 6634097.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9973385334014893, "rewards/meter/std": 1.0115243185282452e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973385334014893, "rewards/total_composite/std": 1.0115243185282452e-06, "reward": 0.9973385334014893, "reward_std": 1.0115243185282452e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0019246861338615417, "sampling/sampling_logp_difference/max": 0.2358248233795166, "sampling/importance_sampling_ratio/min": 0.9778724312782288, "sampling/importance_sampling_ratio/mean": 1.0016769170761108, "sampling/importance_sampling_ratio/max": 1.2659525871276855, "entropy": 0.0171608105301857, "clip_ratio/low_mean": 0.0020491802133619785, "clip_ratio/low_min": 0.0020491802133619785, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0020491802133619785, "reward_total_mean": 0.9973385334014893, "reward_meter_mean": 0.9973385334014893, "reward_meter_std": 1.0115243185282452e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9973385334014893, "reward_total_composite_std": 1.0115243185282452e-06} {"timestamp_utc": "2026-04-12T02:59:53Z", "mode": "train", "global_step": 2925, "epoch": 0.11748403422099048, "loss": 0.0039, "grad_norm": 1.0003553628921509, "learning_rate": 1.1393939393939395e-06, "num_tokens": 6636616.0, "completions/mean_length": 141.875, "completions/min_length": 140.0, "completions/max_length": 143.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 141.875, "completions/min_terminated_length": 140.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.999082088470459, "rewards/meter/std": 0.00011351890861988068, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999082088470459, "rewards/total_composite/std": 0.00011351890861988068, "reward": 0.999082088470459, "reward_std": 0.00011351749708410352, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02242162451148033, "sampling/sampling_logp_difference/max": 1.4046411514282227, "sampling/importance_sampling_ratio/min": 0.24545511603355408, "sampling/importance_sampling_ratio/mean": 1.0028555393218994, "sampling/importance_sampling_ratio/max": 1.6239076852798462, "entropy": 0.1869067270308733, "clip_ratio/low_mean": 0.007904067635536194, "clip_ratio/low_min": 0.007904067635536194, "clip_ratio/high_mean": 0.00530050863744691, "clip_ratio/high_max": 0.00530050863744691, "clip_ratio/region_mean": 0.013204576272983104, "reward_total_mean": 0.999082088470459, "reward_meter_mean": 0.999082088470459, "reward_meter_std": 0.00011351890861988068, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.999082088470459, "reward_total_composite_std": 0.00011351890861988068} {"timestamp_utc": "2026-04-12T02:59:59Z", "mode": "train", "global_step": 2926, "epoch": 0.11752419970277543, "loss": -0.0091, "grad_norm": 3.0424153804779053, "learning_rate": 1.1363636363636364e-06, "num_tokens": 6639784.0, "completions/mean_length": 208.0, "completions/min_length": 201.0, "completions/max_length": 220.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 208.0, "completions/min_terminated_length": 201.0, "completions/max_terminated_length": 220.0, "rewards/meter/mean": 0.8537713289260864, "rewards/meter/std": 0.2598452866077423, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9204545617103577, "rewards/repeat_penalty/std": 0.0758657231926918, "rewards/total_composite/mean": 0.7895516157150269, "rewards/total_composite/std": 0.25156062841415405, "reward": 0.7895516157150269, "reward_std": 0.25156062841415405, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.046991463750600815, "sampling/sampling_logp_difference/max": 1.488713264465332, "sampling/importance_sampling_ratio/min": 0.22566285729408264, "sampling/importance_sampling_ratio/mean": 1.012603998184204, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47782545536756516, "clip_ratio/low_mean": 0.00918415142223239, "clip_ratio/low_min": 0.00918415142223239, "clip_ratio/high_mean": 0.029786661034449935, "clip_ratio/high_max": 0.029786661034449935, "clip_ratio/region_mean": 0.038970812456682324, "reward_total_mean": 0.7895516157150269, "reward_meter_mean": 0.8537713289260864, "reward_meter_std": 0.2598452866077423, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9204545617103577, "reward_repeat_penalty_std": 0.0758657231926918, "reward_total_composite_mean": 0.7895516157150269, "reward_total_composite_std": 0.25156062841415405} {"timestamp_utc": "2026-04-12T03:00:03Z", "mode": "train", "global_step": 2927, "epoch": 0.11756436518456038, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.1333333333333334e-06, "num_tokens": 6641264.0, "completions/mean_length": 32.0, "completions/min_length": 32.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "reward": 0.9992982149124146, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 4.449375410331413e-05, "sampling/sampling_logp_difference/max": 0.00045358005445450544, "sampling/importance_sampling_ratio/min": 0.999942421913147, "sampling/importance_sampling_ratio/mean": 1.0000441074371338, "sampling/importance_sampling_ratio/max": 1.0004537105560303, "entropy": 0.0003692922255140729, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9992982149124146, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:00:10Z", "mode": "train", "global_step": 2928, "epoch": 0.11760453066634534, "loss": 0.003, "grad_norm": 1.2671390771865845, "learning_rate": 1.1303030303030305e-06, "num_tokens": 6644813.0, "completions/mean_length": 236.625, "completions/min_length": 233.0, "completions/max_length": 239.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 236.625, "completions/min_terminated_length": 233.0, "completions/max_terminated_length": 239.0, "rewards/meter/mean": 0.9991750121116638, "rewards/meter/std": 0.00013307879271451384, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991750121116638, "rewards/total_composite/std": 0.00013307879271451384, "reward": 0.9991750121116638, "reward_std": 0.00013307768676895648, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.035225577652454376, "sampling/sampling_logp_difference/max": 1.075662612915039, "sampling/importance_sampling_ratio/min": 0.3410716950893402, "sampling/importance_sampling_ratio/mean": 1.0087391138076782, "sampling/importance_sampling_ratio/max": 1.8521274328231812, "entropy": 0.3537592738866806, "clip_ratio/low_mean": 0.012113503995351493, "clip_ratio/low_min": 0.012113503995351493, "clip_ratio/high_mean": 0.007456094026565552, "clip_ratio/high_max": 0.007456094026565552, "clip_ratio/region_mean": 0.019569598021917045, "reward_total_mean": 0.9991750121116638, "reward_meter_mean": 0.9991750121116638, "reward_meter_std": 0.00013307879271451384, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9991750121116638, "reward_total_composite_std": 0.00013307879271451384} {"timestamp_utc": "2026-04-12T03:00:14Z", "mode": "train", "global_step": 2929, "epoch": 0.11764469614813029, "loss": -0.0, "grad_norm": 0.040891580283641815, "learning_rate": 1.1272727272727275e-06, "num_tokens": 6646628.0, "completions/mean_length": 66.875, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.875, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981529116630554, "rewards/meter/std": 2.428059815429151e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981529116630554, "rewards/total_composite/std": 2.428059815429151e-06, "reward": 0.9981529116630554, "reward_std": 2.428059815429151e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0030419672839343548, "sampling/sampling_logp_difference/max": 0.3047952651977539, "sampling/importance_sampling_ratio/min": 0.7372743487358093, "sampling/importance_sampling_ratio/mean": 1.0010944604873657, "sampling/importance_sampling_ratio/max": 1.0871691703796387, "entropy": 0.029220073018223047, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0018939394503831863, "reward_total_mean": 0.9981529116630554, "reward_meter_mean": 0.9981529116630554, "reward_meter_std": 2.428059815429151e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981529116630554, "reward_total_composite_std": 2.428059815429151e-06} {"timestamp_utc": "2026-04-12T03:00:19Z", "mode": "train", "global_step": 2930, "epoch": 0.11768486162991525, "loss": -0.0005, "grad_norm": 0.1988767832517624, "learning_rate": 1.1242424242424243e-06, "num_tokens": 6648339.0, "completions/mean_length": 66.875, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.875, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.998143196105957, "rewards/meter/std": 1.6269057596218772e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998143196105957, "rewards/total_composite/std": 1.6269057596218772e-05, "reward": 0.998143196105957, "reward_std": 1.6278554539894685e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007081957999616861, "sampling/sampling_logp_difference/max": 0.8161485195159912, "sampling/importance_sampling_ratio/min": 0.44213125109672546, "sampling/importance_sampling_ratio/mean": 0.9996000528335571, "sampling/importance_sampling_ratio/max": 1.115285038948059, "entropy": 0.03760868404060602, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0018656715983524919, "clip_ratio/high_max": 0.0018656715983524919, "clip_ratio/region_mean": 0.0018656715983524919, "reward_total_mean": 0.998143196105957, "reward_meter_mean": 0.998143196105957, "reward_meter_std": 1.6269057596218772e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.998143196105957, "reward_total_composite_std": 1.6269057596218772e-05} {"timestamp_utc": "2026-04-12T03:00:23Z", "mode": "train", "global_step": 2931, "epoch": 0.1177250271117002, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.1212121212121214e-06, "num_tokens": 6650299.0, "completions/mean_length": 80.0, "completions/min_length": 80.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.0, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "reward": 0.7190229296684265, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0029783351346850395, "sampling/sampling_logp_difference/max": 1.665567398071289, "sampling/importance_sampling_ratio/min": 0.18908332288265228, "sampling/importance_sampling_ratio/mean": 0.9990981221199036, "sampling/importance_sampling_ratio/max": 1.0083446502685547, "entropy": 0.002932642324594781, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7190229296684265, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:00:27Z", "mode": "train", "global_step": 2932, "epoch": 0.11776519259348515, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.1181818181818182e-06, "num_tokens": 6651915.0, "completions/mean_length": 29.0, "completions/min_length": 29.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9957436919212341, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957436919212341, "rewards/total_composite/std": 0.0, "reward": 0.9957436919212341, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 7.779466250212863e-05, "sampling/sampling_logp_difference/max": 0.0008399296784773469, "sampling/importance_sampling_ratio/min": 0.9997352361679077, "sampling/importance_sampling_ratio/mean": 1.0000754594802856, "sampling/importance_sampling_ratio/max": 1.000840187072754, "entropy": 0.0006702992213831749, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9957436919212341, "reward_meter_mean": 0.9957436919212341, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9957436919212341, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:00:32Z", "mode": "train", "global_step": 2933, "epoch": 0.11780535807527011, "loss": -0.0002, "grad_norm": 0.009405174292623997, "learning_rate": 1.1151515151515153e-06, "num_tokens": 6653707.0, "completions/mean_length": 67.0, "completions/min_length": 67.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981515407562256, "rewards/meter/std": 1.501989686403249e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981515407562256, "rewards/total_composite/std": 1.501989686403249e-06, "reward": 0.9981515407562256, "reward_std": 1.5057863720358e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0026725942734628916, "sampling/sampling_logp_difference/max": 0.09437625110149384, "sampling/importance_sampling_ratio/min": 0.9517659544944763, "sampling/importance_sampling_ratio/mean": 1.002063512802124, "sampling/importance_sampling_ratio/max": 1.0989731550216675, "entropy": 0.028828308917582035, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0018656715983524919, "reward_total_mean": 0.9981515407562256, "reward_meter_mean": 0.9981515407562256, "reward_meter_std": 1.501989686403249e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981515407562256, "reward_total_composite_std": 1.501989686403249e-06} {"timestamp_utc": "2026-04-12T03:00:36Z", "mode": "train", "global_step": 2934, "epoch": 0.11784552355705506, "loss": -0.0064, "grad_norm": 2.1487016677856445, "learning_rate": 1.112121212121212e-06, "num_tokens": 6655568.0, "completions/mean_length": 78.625, "completions/min_length": 77.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.625, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9986746311187744, "rewards/meter/std": 0.0008188642095774412, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9986746311187744, "rewards/total_composite/std": 0.0008188642095774412, "reward": 0.9986746311187744, "reward_std": 0.0008188657811842859, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020405031740665436, "sampling/sampling_logp_difference/max": 0.8371596336364746, "sampling/importance_sampling_ratio/min": 0.43293848633766174, "sampling/importance_sampling_ratio/mean": 1.002895712852478, "sampling/importance_sampling_ratio/max": 1.461742877960205, "entropy": 0.18547690846025944, "clip_ratio/low_mean": 0.001623376621864736, "clip_ratio/low_min": 0.001623376621864736, "clip_ratio/high_mean": 0.01744664926081896, "clip_ratio/high_max": 0.01744664926081896, "clip_ratio/region_mean": 0.019070025882683694, "reward_total_mean": 0.9986746311187744, "reward_meter_mean": 0.9986746311187744, "reward_meter_std": 0.0008188642095774412, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9986746311187744, "reward_total_composite_std": 0.0008188642095774412} {"timestamp_utc": "2026-04-12T03:00:41Z", "mode": "train", "global_step": 2935, "epoch": 0.11788568903884002, "loss": 0.0009, "grad_norm": 0.6610411405563354, "learning_rate": 1.1090909090909093e-06, "num_tokens": 6657678.0, "completions/mean_length": 92.75, "completions/min_length": 92.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.75, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.9977371692657471, "rewards/meter/std": 7.424020441249013e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977371692657471, "rewards/total_composite/std": 7.424020441249013e-05, "reward": 0.9977371692657471, "reward_std": 7.424017530865967e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011522673070430756, "sampling/sampling_logp_difference/max": 0.7292871475219727, "sampling/importance_sampling_ratio/min": 0.48225268721580505, "sampling/importance_sampling_ratio/mean": 1.0004804134368896, "sampling/importance_sampling_ratio/max": 1.3887077569961548, "entropy": 0.09037415590137243, "clip_ratio/low_mean": 0.005390953738242388, "clip_ratio/low_min": 0.005390953738242388, "clip_ratio/high_mean": 0.008094357093796134, "clip_ratio/high_max": 0.008094357093796134, "clip_ratio/region_mean": 0.013485310832038522, "reward_total_mean": 0.9977371692657471, "reward_meter_mean": 0.9977371692657471, "reward_meter_std": 7.424020441249013e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9977371692657471, "reward_total_composite_std": 7.424020441249013e-05} {"timestamp_utc": "2026-04-12T03:00:51Z", "mode": "train", "global_step": 2936, "epoch": 0.11792585452062497, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.1060606060606062e-06, "num_tokens": 6659326.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9992560148239136, "rewards/meter/std": 9.200150816468522e-05, "rewards/count_adherence/mean": 0.6499999761581421, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9709615111351013, "rewards/repeat_penalty/std": 0.027279434725642204, "rewards/total_composite/mean": 0.6306551694869995, "rewards/total_composite/std": 0.01770690269768238, "reward": 0.6306551694869995, "reward_std": 0.017706912010908127, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.6306551694869995, "reward_meter_mean": 0.9992560148239136, "reward_meter_std": 9.200150816468522e-05, "reward_count_adherence_mean": 0.6499999761581421, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9709615111351013, "reward_repeat_penalty_std": 0.027279434725642204, "reward_total_composite_mean": 0.6306551694869995, "reward_total_composite_std": 0.01770690269768238} {"timestamp_utc": "2026-04-12T03:00:56Z", "mode": "train", "global_step": 2937, "epoch": 0.11796602000240992, "loss": 0.0054, "grad_norm": 1.2551523447036743, "learning_rate": 1.1030303030303032e-06, "num_tokens": 6661254.0, "completions/mean_length": 79.0, "completions/min_length": 78.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.0, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9990774989128113, "rewards/meter/std": 0.00010208567255176604, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990774989128113, "rewards/total_composite/std": 0.00010208567255176604, "reward": 0.9990774989128113, "reward_std": 0.00010208544699708, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01773812435567379, "sampling/sampling_logp_difference/max": 0.474831223487854, "sampling/importance_sampling_ratio/min": 0.6219900250434875, "sampling/importance_sampling_ratio/mean": 1.0067074298858643, "sampling/importance_sampling_ratio/max": 1.4563788175582886, "entropy": 0.16411693580448627, "clip_ratio/low_mean": 0.007854048046283424, "clip_ratio/low_min": 0.007854048046283424, "clip_ratio/high_mean": 0.006369685288518667, "clip_ratio/high_max": 0.006369685288518667, "clip_ratio/region_mean": 0.014223733334802091, "reward_total_mean": 0.9990774989128113, "reward_meter_mean": 0.9990774989128113, "reward_meter_std": 0.00010208567255176604, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990774989128113, "reward_total_composite_std": 0.00010208567255176604} {"timestamp_utc": "2026-04-12T03:01:00Z", "mode": "train", "global_step": 2938, "epoch": 0.11800618548419488, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.1e-06, "num_tokens": 6662614.0, "completions/mean_length": 32.0, "completions/min_length": 32.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "reward": 0.9992982149124146, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 9.101704199565575e-05, "sampling/sampling_logp_difference/max": 0.003561503253877163, "sampling/importance_sampling_ratio/min": 0.9964448809623718, "sampling/importance_sampling_ratio/mean": 1.00003981590271, "sampling/importance_sampling_ratio/max": 1.001272439956665, "entropy": 0.0009229831339325756, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9992982149124146, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:01:04Z", "mode": "train", "global_step": 2939, "epoch": 0.11804635096597983, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.096969696969697e-06, "num_tokens": 6664078.0, "completions/mean_length": 35.0, "completions/min_length": 35.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9995916485786438, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9995916485786438, "rewards/total_composite/std": 0.0, "reward": 0.9995916485786438, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.002187325619161129, "sampling/sampling_logp_difference/max": 0.01813378371298313, "sampling/importance_sampling_ratio/min": 0.9820296168327332, "sampling/importance_sampling_ratio/mean": 1.001636266708374, "sampling/importance_sampling_ratio/max": 1.0143886804580688, "entropy": 0.023612842429429293, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9995916485786438, "reward_meter_mean": 0.9995916485786438, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9995916485786438, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:01:12Z", "mode": "train", "global_step": 2940, "epoch": 0.11808651644776479, "loss": 0.0184, "grad_norm": 1.6798667907714844, "learning_rate": 1.093939393939394e-06, "num_tokens": 6668028.0, "completions/mean_length": 290.75, "completions/min_length": 284.0, "completions/max_length": 304.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 290.75, "completions/min_terminated_length": 284.0, "completions/max_terminated_length": 304.0, "rewards/meter/mean": 0.9990341663360596, "rewards/meter/std": 9.884726023301482e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9598958492279053, "rewards/repeat_penalty/std": 0.04713716357946396, "rewards/total_composite/mean": 0.9589698314666748, "rewards/total_composite/std": 0.04711604490876198, "reward": 0.9589698314666748, "reward_std": 0.04711604863405228, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026417069137096405, "sampling/sampling_logp_difference/max": 1.65692138671875, "sampling/importance_sampling_ratio/min": 0.19072523713111877, "sampling/importance_sampling_ratio/mean": 1.0048670768737793, "sampling/importance_sampling_ratio/max": 1.720551609992981, "entropy": 0.299512580037117, "clip_ratio/low_mean": 0.007625087164342403, "clip_ratio/low_min": 0.007625087164342403, "clip_ratio/high_mean": 0.009173946105875075, "clip_ratio/high_max": 0.009173946105875075, "clip_ratio/region_mean": 0.01679903327021748, "reward_total_mean": 0.9589698314666748, "reward_meter_mean": 0.9990341663360596, "reward_meter_std": 9.884726023301482e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9598958492279053, "reward_repeat_penalty_std": 0.04713716357946396, "reward_total_composite_mean": 0.9589698314666748, "reward_total_composite_std": 0.04711604490876198} {"timestamp_utc": "2026-04-12T03:01:18Z", "mode": "train", "global_step": 2941, "epoch": 0.11812668192954974, "loss": 0.0035, "grad_norm": 1.4318662881851196, "learning_rate": 1.090909090909091e-06, "num_tokens": 6670786.0, "completions/mean_length": 165.75, "completions/min_length": 164.0, "completions/max_length": 167.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 165.75, "completions/min_terminated_length": 164.0, "completions/max_terminated_length": 167.0, "rewards/meter/mean": 0.9992879629135132, "rewards/meter/std": 0.0003045983612537384, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.9715272188186646, "rewards/total_composite/std": 0.05134394392371178, "reward": 0.9715272188186646, "reward_std": 0.05134394019842148, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015097087249159813, "sampling/sampling_logp_difference/max": 1.1712437868118286, "sampling/importance_sampling_ratio/min": 0.3099811375141144, "sampling/importance_sampling_ratio/mean": 1.0033372640609741, "sampling/importance_sampling_ratio/max": 1.513466238975525, "entropy": 0.11971556767821312, "clip_ratio/low_mean": 0.003765060333535075, "clip_ratio/low_min": 0.003765060333535075, "clip_ratio/high_mean": 0.009821378625929356, "clip_ratio/high_max": 0.009821378625929356, "clip_ratio/region_mean": 0.01358643895946443, "reward_total_mean": 0.9715272188186646, "reward_meter_mean": 0.9992879629135132, "reward_meter_std": 0.0003045983612537384, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_total_composite_mean": 0.9715272188186646, "reward_total_composite_std": 0.05134394392371178} {"timestamp_utc": "2026-04-12T03:01:25Z", "mode": "train", "global_step": 2942, "epoch": 0.1181668474113347, "loss": -0.0003, "grad_norm": 2.797619581222534, "learning_rate": 1.087878787878788e-06, "num_tokens": 6674569.0, "completions/mean_length": 278.875, "completions/min_length": 275.0, "completions/max_length": 282.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 278.875, "completions/min_terminated_length": 275.0, "completions/max_terminated_length": 282.0, "rewards/meter/mean": 0.9974960684776306, "rewards/meter/std": 0.000377756921807304, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9338235855102539, "rewards/repeat_penalty/std": 0.020797256380319595, "rewards/total_composite/mean": 0.8150506019592285, "rewards/total_composite/std": 0.018201656639575958, "reward": 0.8150506019592285, "reward_std": 0.018201671540737152, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03948317840695381, "sampling/sampling_logp_difference/max": 1.47027587890625, "sampling/importance_sampling_ratio/min": 0.22986207902431488, "sampling/importance_sampling_ratio/mean": 1.0059531927108765, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31575205735862255, "clip_ratio/low_mean": 0.0027173913549631834, "clip_ratio/low_min": 0.0027173913549631834, "clip_ratio/high_mean": 0.026906014885753393, "clip_ratio/high_max": 0.026906014885753393, "clip_ratio/region_mean": 0.029623406240716577, "reward_total_mean": 0.8150506019592285, "reward_meter_mean": 0.9974960684776306, "reward_meter_std": 0.000377756921807304, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9338235855102539, "reward_repeat_penalty_std": 0.020797256380319595, "reward_total_composite_mean": 0.8150506019592285, "reward_total_composite_std": 0.018201656639575958} {"timestamp_utc": "2026-04-12T03:01:32Z", "mode": "train", "global_step": 2943, "epoch": 0.11820701289311965, "loss": 0.0233, "grad_norm": 3.1147727966308594, "learning_rate": 1.084848484848485e-06, "num_tokens": 6678237.0, "completions/mean_length": 271.5, "completions/min_length": 258.0, "completions/max_length": 284.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 271.5, "completions/min_terminated_length": 258.0, "completions/max_terminated_length": 284.0, "rewards/meter/mean": 0.9941989779472351, "rewards/meter/std": 0.0028337803669273853, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9489378929138184, "rewards/repeat_penalty/std": 0.02066386677324772, "rewards/total_composite/mean": 0.8254706859588623, "rewards/total_composite/std": 0.016226978972554207, "reward": 0.8254706859588623, "reward_std": 0.0162269975990057, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.049873415380716324, "sampling/sampling_logp_difference/max": 3.8973119258880615, "sampling/importance_sampling_ratio/min": 0.02029639668762684, "sampling/importance_sampling_ratio/mean": 1.0050019025802612, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37596092373132706, "clip_ratio/low_mean": 0.02794346760492772, "clip_ratio/low_min": 0.02794346760492772, "clip_ratio/high_mean": 0.006298449821770191, "clip_ratio/high_max": 0.006298449821770191, "clip_ratio/region_mean": 0.03424191742669791, "reward_total_mean": 0.8254706859588623, "reward_meter_mean": 0.9941989779472351, "reward_meter_std": 0.0028337803669273853, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9489378929138184, "reward_repeat_penalty_std": 0.02066386677324772, "reward_total_composite_mean": 0.8254706859588623, "reward_total_composite_std": 0.016226978972554207} {"timestamp_utc": "2026-04-12T03:01:37Z", "mode": "train", "global_step": 2944, "epoch": 0.1182471783749046, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.081818181818182e-06, "num_tokens": 6680078.0, "completions/mean_length": 67.125, "completions/min_length": 67.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.125, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9981522560119629, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981522560119629, "rewards/total_composite/std": 0.0, "reward": 0.9981522560119629, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0032630774658173323, "sampling/sampling_logp_difference/max": 0.3072841167449951, "sampling/importance_sampling_ratio/min": 0.946239173412323, "sampling/importance_sampling_ratio/mean": 1.0029489994049072, "sampling/importance_sampling_ratio/max": 1.359727144241333, "entropy": 0.027955329976975918, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9981522560119629, "reward_meter_mean": 0.9981522560119629, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981522560119629, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:01:41Z", "mode": "train", "global_step": 2945, "epoch": 0.11828734385668956, "loss": -0.0019, "grad_norm": 1.8361525535583496, "learning_rate": 1.078787878787879e-06, "num_tokens": 6681590.0, "completions/mean_length": 68.0, "completions/min_length": 67.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9993088841438293, "rewards/meter/std": 0.0001494528987677768, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993088841438293, "rewards/total_composite/std": 0.0001494528987677768, "reward": 0.9993088841438293, "reward_std": 0.0001494480820838362, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01226960588246584, "sampling/sampling_logp_difference/max": 0.6763467788696289, "sampling/importance_sampling_ratio/min": 0.5084711313247681, "sampling/importance_sampling_ratio/mean": 1.0019557476043701, "sampling/importance_sampling_ratio/max": 1.2217551469802856, "entropy": 0.08658721391111612, "clip_ratio/low_mean": 0.01105684821959585, "clip_ratio/low_min": 0.01105684821959585, "clip_ratio/high_mean": 0.0018382353009656072, "clip_ratio/high_max": 0.0018382353009656072, "clip_ratio/region_mean": 0.012895083520561457, "reward_total_mean": 0.9993088841438293, "reward_meter_mean": 0.9993088841438293, "reward_meter_std": 0.0001494528987677768, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993088841438293, "reward_total_composite_std": 0.0001494528987677768} {"timestamp_utc": "2026-04-12T03:01:46Z", "mode": "train", "global_step": 2946, "epoch": 0.11832750933847451, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.0757575757575758e-06, "num_tokens": 6683270.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0003203663509339094, "sampling/sampling_logp_difference/max": 0.004554374143481255, "sampling/importance_sampling_ratio/min": 0.9995915293693542, "sampling/importance_sampling_ratio/mean": 1.0003174543380737, "sampling/importance_sampling_ratio/max": 1.0045647621154785, "entropy": 0.0023213086824398488, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:01:51Z", "mode": "train", "global_step": 2947, "epoch": 0.11836767482025946, "loss": 0.0031, "grad_norm": 3.6917295455932617, "learning_rate": 1.0727272727272728e-06, "num_tokens": 6685877.0, "completions/mean_length": 118.875, "completions/min_length": 115.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.875, "completions/min_terminated_length": 115.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.9956615567207336, "rewards/meter/std": 0.0021291025914251804, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9285714626312256, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.9246125221252441, "rewards/total_composite/std": 0.07701455056667328, "reward": 0.9246125221252441, "reward_std": 0.07701455056667328, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03363678976893425, "sampling/sampling_logp_difference/max": 1.5754187107086182, "sampling/importance_sampling_ratio/min": 0.20692089200019836, "sampling/importance_sampling_ratio/mean": 1.0035653114318848, "sampling/importance_sampling_ratio/max": 1.86605966091156, "entropy": 0.20067131519317627, "clip_ratio/low_mean": 0.02329266769811511, "clip_ratio/low_min": 0.02329266769811511, "clip_ratio/high_mean": 0.015964585822075605, "clip_ratio/high_max": 0.015964585822075605, "clip_ratio/region_mean": 0.039257253520190716, "reward_total_mean": 0.9246125221252441, "reward_meter_mean": 0.9956615567207336, "reward_meter_std": 0.0021291025914251804, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9285714626312256, "reward_repeat_penalty_std": 0.07636035233736038, "reward_total_composite_mean": 0.9246125221252441, "reward_total_composite_std": 0.07701455056667328} {"timestamp_utc": "2026-04-12T03:01:56Z", "mode": "train", "global_step": 2948, "epoch": 0.11840784030204442, "loss": 0.0056, "grad_norm": 3.9713573455810547, "learning_rate": 1.0696969696969696e-06, "num_tokens": 6687719.0, "completions/mean_length": 69.25, "completions/min_length": 66.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.25, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9902843236923218, "rewards/meter/std": 0.01087514590471983, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9902843236923218, "rewards/total_composite/std": 0.01087514590471983, "reward": 0.9902843236923218, "reward_std": 0.010875147767364979, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.040240850299596786, "sampling/sampling_logp_difference/max": 1.782811164855957, "sampling/importance_sampling_ratio/min": 0.16816474497318268, "sampling/importance_sampling_ratio/mean": 0.9977670311927795, "sampling/importance_sampling_ratio/max": 1.6766802072525024, "entropy": 0.31914436258375645, "clip_ratio/low_mean": 0.007194617064669728, "clip_ratio/low_min": 0.007194617064669728, "clip_ratio/high_mean": 0.01954207243397832, "clip_ratio/high_max": 0.01954207243397832, "clip_ratio/region_mean": 0.026736689498648047, "reward_total_mean": 0.9902843236923218, "reward_meter_mean": 0.9902843236923218, "reward_meter_std": 0.01087514590471983, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9902843236923218, "reward_total_composite_std": 0.01087514590471983} {"timestamp_utc": "2026-04-12T03:02:02Z", "mode": "train", "global_step": 2949, "epoch": 0.11844800578382937, "loss": -0.001, "grad_norm": 2.246262311935425, "learning_rate": 1.066666666666667e-06, "num_tokens": 6690261.0, "completions/mean_length": 128.75, "completions/min_length": 128.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.75, "completions/min_terminated_length": 128.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9919617176055908, "rewards/meter/std": 0.01668241061270237, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9919617176055908, "rewards/total_composite/std": 0.01668241061270237, "reward": 0.9919617176055908, "reward_std": 0.01668240688741207, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017122527584433556, "sampling/sampling_logp_difference/max": 1.4149751663208008, "sampling/importance_sampling_ratio/min": 0.24293164908885956, "sampling/importance_sampling_ratio/mean": 0.999458909034729, "sampling/importance_sampling_ratio/max": 1.6150113344192505, "entropy": 0.15123706310987473, "clip_ratio/low_mean": 0.0019379844889044762, "clip_ratio/low_min": 0.0019379844889044762, "clip_ratio/high_mean": 0.01359617244452238, "clip_ratio/high_max": 0.01359617244452238, "clip_ratio/region_mean": 0.015534156933426857, "reward_total_mean": 0.9919617176055908, "reward_meter_mean": 0.9919617176055908, "reward_meter_std": 0.01668241061270237, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9919617176055908, "reward_total_composite_std": 0.01668241061270237} {"timestamp_utc": "2026-04-12T03:02:07Z", "mode": "train", "global_step": 2950, "epoch": 0.11848817126561433, "loss": -0.0107, "grad_norm": 4.775022029876709, "learning_rate": 1.0636363636363637e-06, "num_tokens": 6692334.0, "completions/mean_length": 90.125, "completions/min_length": 88.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.125, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.9957476258277893, "rewards/meter/std": 0.0016542106168344617, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.8962882161140442, "rewards/total_composite/std": 0.10756382346153259, "reward": 0.8962882161140442, "reward_std": 0.107563816010952, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03922773897647858, "sampling/sampling_logp_difference/max": 3.326974630355835, "sampling/importance_sampling_ratio/min": 0.03590155765414238, "sampling/importance_sampling_ratio/mean": 0.9992499947547913, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.16245986334979534, "clip_ratio/low_mean": 0.009927221923135221, "clip_ratio/low_min": 0.009927221923135221, "clip_ratio/high_mean": 0.009542036801576614, "clip_ratio/high_max": 0.009542036801576614, "clip_ratio/region_mean": 0.019469258724711835, "reward_total_mean": 0.8962882161140442, "reward_meter_mean": 0.9957476258277893, "reward_meter_std": 0.0016542106168344617, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_total_composite_mean": 0.8962882161140442, "reward_total_composite_std": 0.10756382346153259} {"timestamp_utc": "2026-04-12T03:03:22Z", "mode": "eval", "global_step": 2950, "epoch": 0.11848817126561433, "eval_loss": NaN, "eval_runtime": 75.4625, "eval_samples_per_second": 1.378, "eval_steps_per_second": 0.172, "eval_num_tokens": 6692334.0, "eval_completions/mean_length": 208.25, "eval_completions/min_length": 60.15384615384615, "eval_completions/max_length": 403.61538461538464, "eval_completions/clipped_ratio": 0.009615384615384616, "eval_completions/mean_terminated_length": 205.135989849384, "eval_completions/min_terminated_length": 60.15384615384615, "eval_completions/max_terminated_length": 395.0769230769231, "eval_rewards/meter/mean": 0.7935946950545678, "eval_rewards/meter/std": 0.32434708754030556, "eval_rewards/count_adherence/mean": 0.9516645761636587, "eval_rewards/count_adherence/std": 0.06901963714223641, "eval_rewards/arabic_clean/mean": 0.9711538461538461, "eval_rewards/arabic_clean/std": 0.06280488005051246, "eval_rewards/repeat_penalty/mean": 0.9551343642748319, "eval_rewards/repeat_penalty/std": 0.08046436768311721, "eval_rewards/total_composite/mean": 0.7131056900207813, "eval_rewards/total_composite/std": 0.3318366717833739, "eval_reward": 0.7131056900207813, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.032634207692283854, "eval_sampling/sampling_logp_difference/max": 1.3049519062042236, "eval_sampling/importance_sampling_ratio/min": 0.27521977172448087, "eval_sampling/importance_sampling_ratio/mean": 1.0077741604584913, "eval_sampling/importance_sampling_ratio/max": 1.446980045391963, "eval_entropy": 0.38887230020302993, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7131056900207813, "eval_reward_meter_mean": 0.7935946950545678, "eval_reward_meter_std": 0.32434708754030556, "eval_reward_count_adherence_mean": 0.9516645761636587, "eval_reward_count_adherence_std": 0.06901963714223641, "eval_reward_arabic_clean_mean": 0.9711538461538461, "eval_reward_arabic_clean_std": 0.06280488005051246, "eval_reward_repeat_penalty_mean": 0.9551343642748319, "eval_reward_repeat_penalty_std": 0.08046436768311721, "eval_reward_total_composite_mean": 0.7131056900207813, "eval_reward_total_composite_std": 0.3318366717833739} {"timestamp_utc": "2026-04-12T03:03:29Z", "mode": "train", "global_step": 2951, "epoch": 0.11852833674739928, "loss": -0.0003, "grad_norm": 7.092353820800781, "learning_rate": 1.0606060606060608e-06, "num_tokens": 6693891.0, "completions/mean_length": 33.625, "completions/min_length": 32.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.625, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9904308915138245, "rewards/meter/std": 0.004542448092252016, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9904308915138245, "rewards/total_composite/std": 0.004542448092252016, "reward": 0.9904308915138245, "reward_std": 0.004542447626590729, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021938247606158257, "sampling/sampling_logp_difference/max": 1.2846660614013672, "sampling/importance_sampling_ratio/min": 0.27674299478530884, "sampling/importance_sampling_ratio/mean": 1.0013483762741089, "sampling/importance_sampling_ratio/max": 1.398740530014038, "entropy": 0.1007662764750421, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0037878789007663727, "reward_total_mean": 0.9904308915138245, "reward_meter_mean": 0.9904308915138245, "reward_meter_std": 0.004542448092252016, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9904308915138245, "reward_total_composite_std": 0.004542448092252016} {"timestamp_utc": "2026-04-12T03:03:34Z", "mode": "train", "global_step": 2952, "epoch": 0.11856850222918423, "loss": 0.0018, "grad_norm": 0.5699337124824524, "learning_rate": 1.0575757575757576e-06, "num_tokens": 6695487.0, "completions/mean_length": 36.5, "completions/min_length": 36.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.5, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.9995974898338318, "rewards/meter/std": 1.9893313947250135e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9995974898338318, "rewards/total_composite/std": 1.9893313947250135e-05, "reward": 0.9995974898338318, "reward_std": 1.9893312128260732e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009155918844044209, "sampling/sampling_logp_difference/max": 0.8658480644226074, "sampling/importance_sampling_ratio/min": 0.42069464921951294, "sampling/importance_sampling_ratio/mean": 0.9967590570449829, "sampling/importance_sampling_ratio/max": 1.0766030550003052, "entropy": 0.03357855952344835, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.013795045204460621, "clip_ratio/high_max": 0.013795045204460621, "clip_ratio/region_mean": 0.013795045204460621, "reward_total_mean": 0.9995974898338318, "reward_meter_mean": 0.9995974898338318, "reward_meter_std": 1.9893313947250135e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9995974898338318, "reward_total_composite_std": 1.9893313947250135e-05} {"timestamp_utc": "2026-04-12T03:03:39Z", "mode": "train", "global_step": 2953, "epoch": 0.11860866771096919, "loss": -0.0001, "grad_norm": 1.022800087928772, "learning_rate": 1.0545454545454547e-06, "num_tokens": 6697761.0, "completions/mean_length": 107.25, "completions/min_length": 106.0, "completions/max_length": 110.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.25, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 110.0, "rewards/meter/mean": 0.9992892742156982, "rewards/meter/std": 9.131882688961923e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992892742156982, "rewards/total_composite/std": 9.131882688961923e-05, "reward": 0.9992892742156982, "reward_std": 9.13253243197687e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016012487933039665, "sampling/sampling_logp_difference/max": 0.9576692581176758, "sampling/importance_sampling_ratio/min": 0.3837863802909851, "sampling/importance_sampling_ratio/mean": 1.0017812252044678, "sampling/importance_sampling_ratio/max": 1.6504535675048828, "entropy": 0.13113553449511528, "clip_ratio/low_mean": 0.002336448524147272, "clip_ratio/low_min": 0.002336448524147272, "clip_ratio/high_mean": 0.00930431904271245, "clip_ratio/high_max": 0.00930431904271245, "clip_ratio/region_mean": 0.011640767566859722, "reward_total_mean": 0.9992892742156982, "reward_meter_mean": 0.9992892742156982, "reward_meter_std": 9.131882688961923e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992892742156982, "reward_total_composite_std": 9.131882688961923e-05} {"timestamp_utc": "2026-04-12T03:03:44Z", "mode": "train", "global_step": 2954, "epoch": 0.11864883319275414, "loss": -0.0034, "grad_norm": 10.395458221435547, "learning_rate": 1.0515151515151515e-06, "num_tokens": 6699386.0, "completions/mean_length": 42.125, "completions/min_length": 40.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.125, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.9281213283538818, "rewards/meter/std": 0.031521886587142944, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9281213283538818, "rewards/total_composite/std": 0.031521886587142944, "reward": 0.9281213283538818, "reward_std": 0.03152187913656235, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08637839555740356, "sampling/sampling_logp_difference/max": 2.8028721809387207, "sampling/importance_sampling_ratio/min": 0.060635656118392944, "sampling/importance_sampling_ratio/mean": 0.9940267205238342, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.30831943079829216, "clip_ratio/low_mean": 0.024695121683180332, "clip_ratio/low_min": 0.024695121683180332, "clip_ratio/high_mean": 0.04968656366690993, "clip_ratio/high_max": 0.04968656366690993, "clip_ratio/region_mean": 0.07438168535009027, "reward_total_mean": 0.9281213283538818, "reward_meter_mean": 0.9281213283538818, "reward_meter_std": 0.031521886587142944, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9281213283538818, "reward_total_composite_std": 0.031521886587142944} {"timestamp_utc": "2026-04-12T03:03:50Z", "mode": "train", "global_step": 2955, "epoch": 0.1186889986745391, "loss": 0.0177, "grad_norm": 3.04614520072937, "learning_rate": 1.0484848484848485e-06, "num_tokens": 6702346.0, "completions/mean_length": 174.0, "completions/min_length": 166.0, "completions/max_length": 180.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 174.0, "completions/min_terminated_length": 166.0, "completions/max_terminated_length": 180.0, "rewards/meter/mean": 0.9902915358543396, "rewards/meter/std": 0.007541172672063112, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9902915358543396, "rewards/total_composite/std": 0.007541172672063112, "reward": 0.9902915358543396, "reward_std": 0.007541188970208168, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04366350546479225, "sampling/sampling_logp_difference/max": 1.370412826538086, "sampling/importance_sampling_ratio/min": 0.25400206446647644, "sampling/importance_sampling_ratio/mean": 1.0090171098709106, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3894418552517891, "clip_ratio/low_mean": 0.00557116128038615, "clip_ratio/low_min": 0.00557116128038615, "clip_ratio/high_mean": 0.022244701394811273, "clip_ratio/high_max": 0.022244701394811273, "clip_ratio/region_mean": 0.027815862675197423, "reward_total_mean": 0.9902915358543396, "reward_meter_mean": 0.9902915358543396, "reward_meter_std": 0.007541172672063112, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9902915358543396, "reward_total_composite_std": 0.007541172672063112} {"timestamp_utc": "2026-04-12T03:03:55Z", "mode": "train", "global_step": 2956, "epoch": 0.11872916415632405, "loss": 0.0014, "grad_norm": 2.19303035736084, "learning_rate": 1.0454545454545456e-06, "num_tokens": 6704123.0, "completions/mean_length": 68.125, "completions/min_length": 68.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.125, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9994041919708252, "rewards/meter/std": 0.0001539249497000128, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994041919708252, "rewards/total_composite/std": 0.0001539249497000128, "reward": 0.9994041919708252, "reward_std": 0.0001539262884762138, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012835102155804634, "sampling/sampling_logp_difference/max": 1.0354909896850586, "sampling/importance_sampling_ratio/min": 0.35505202412605286, "sampling/importance_sampling_ratio/mean": 1.0002238750457764, "sampling/importance_sampling_ratio/max": 1.366014838218689, "entropy": 0.0865520965307951, "clip_ratio/low_mean": 0.007299659075215459, "clip_ratio/low_min": 0.007299659075215459, "clip_ratio/high_mean": 0.0018382353009656072, "clip_ratio/high_max": 0.0018382353009656072, "clip_ratio/region_mean": 0.009137894376181066, "reward_total_mean": 0.9994041919708252, "reward_meter_mean": 0.9994041919708252, "reward_meter_std": 0.0001539249497000128, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994041919708252, "reward_total_composite_std": 0.0001539249497000128} {"timestamp_utc": "2026-04-12T03:04:00Z", "mode": "train", "global_step": 2957, "epoch": 0.118769329638109, "loss": 0.0006, "grad_norm": 0.0930638462305069, "learning_rate": 1.0424242424242426e-06, "num_tokens": 6705889.0, "completions/mean_length": 66.75, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.75, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.998155951499939, "rewards/meter/std": 9.95435857475968e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998155951499939, "rewards/total_composite/std": 9.95435857475968e-06, "reward": 0.998155951499939, "reward_std": 9.94737092696596e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008290099911391735, "sampling/sampling_logp_difference/max": 0.8198513984680176, "sampling/importance_sampling_ratio/min": 0.44049713015556335, "sampling/importance_sampling_ratio/mean": 0.9991642832756042, "sampling/importance_sampling_ratio/max": 1.253778100013733, "entropy": 0.04649371886625886, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0056535504991188645, "clip_ratio/high_max": 0.0056535504991188645, "clip_ratio/region_mean": 0.0056535504991188645, "reward_total_mean": 0.998155951499939, "reward_meter_mean": 0.998155951499939, "reward_meter_std": 9.95435857475968e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.998155951499939, "reward_total_composite_std": 9.95435857475968e-06} {"timestamp_utc": "2026-04-12T03:04:08Z", "mode": "train", "global_step": 2958, "epoch": 0.11880949511989396, "loss": -0.0383, "grad_norm": 2.60581636428833, "learning_rate": 1.0393939393939394e-06, "num_tokens": 6710000.0, "completions/mean_length": 319.875, "completions/min_length": 293.0, "completions/max_length": 348.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 319.875, "completions/min_terminated_length": 293.0, "completions/max_terminated_length": 348.0, "rewards/meter/mean": 0.9985268712043762, "rewards/meter/std": 0.0011142482981085777, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.0534522607922554, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9794891476631165, "rewards/repeat_penalty/std": 0.028372056782245636, "rewards/total_composite/mean": 0.9288632869720459, "rewards/total_composite/std": 0.05354485660791397, "reward": 0.9288632869720459, "reward_std": 0.05354484170675278, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0562576949596405, "sampling/sampling_logp_difference/max": 1.9502592086791992, "sampling/importance_sampling_ratio/min": 0.14223718643188477, "sampling/importance_sampling_ratio/mean": 1.0072740316390991, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47900502756237984, "clip_ratio/low_mean": 0.016178023535758257, "clip_ratio/low_min": 0.016178023535758257, "clip_ratio/high_mean": 0.023225491400808096, "clip_ratio/high_max": 0.023225491400808096, "clip_ratio/region_mean": 0.03940351493656635, "reward_total_mean": 0.9288632869720459, "reward_meter_mean": 0.9985268712043762, "reward_meter_std": 0.0011142482981085777, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.0534522607922554, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9794891476631165, "reward_repeat_penalty_std": 0.028372056782245636, "reward_total_composite_mean": 0.9288632869720459, "reward_total_composite_std": 0.05354485660791397} {"timestamp_utc": "2026-04-12T03:04:16Z", "mode": "train", "global_step": 2959, "epoch": 0.11884966060167891, "loss": -0.0149, "grad_norm": 1.92487370967865, "learning_rate": 1.0363636363636365e-06, "num_tokens": 6714601.0, "completions/mean_length": 358.125, "completions/min_length": 348.0, "completions/max_length": 382.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 358.125, "completions/min_terminated_length": 348.0, "completions/max_terminated_length": 382.0, "rewards/meter/mean": 0.9991495013237, "rewards/meter/std": 0.0001215613738168031, "rewards/count_adherence/mean": 0.8295454978942871, "rewards/count_adherence/std": 0.03214123100042343, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.985702633857727, "rewards/repeat_penalty/std": 0.02648802287876606, "rewards/total_composite/mean": 0.8171521425247192, "rewards/total_composite/std": 0.04246782884001732, "reward": 0.8171521425247192, "reward_std": 0.042467836290597916, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04391275718808174, "sampling/sampling_logp_difference/max": 1.476973533630371, "sampling/importance_sampling_ratio/min": 0.228327676653862, "sampling/importance_sampling_ratio/mean": 1.0104491710662842, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40936222299933434, "clip_ratio/low_mean": 0.004202505107969046, "clip_ratio/low_min": 0.004202505107969046, "clip_ratio/high_mean": 0.022940220427699387, "clip_ratio/high_max": 0.022940220427699387, "clip_ratio/region_mean": 0.027142725535668433, "reward_total_mean": 0.8171521425247192, "reward_meter_mean": 0.9991495013237, "reward_meter_std": 0.0001215613738168031, "reward_count_adherence_mean": 0.8295454978942871, "reward_count_adherence_std": 0.03214123100042343, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.985702633857727, "reward_repeat_penalty_std": 0.02648802287876606, "reward_total_composite_mean": 0.8171521425247192, "reward_total_composite_std": 0.04246782884001732} {"timestamp_utc": "2026-04-12T03:04:24Z", "mode": "train", "global_step": 2960, "epoch": 0.11888982608346386, "loss": -0.0038, "grad_norm": 2.4538395404815674, "learning_rate": 1.0333333333333333e-06, "num_tokens": 6719098.0, "completions/mean_length": 355.125, "completions/min_length": 340.0, "completions/max_length": 360.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 355.125, "completions/min_terminated_length": 340.0, "completions/max_terminated_length": 360.0, "rewards/meter/mean": 0.9989354610443115, "rewards/meter/std": 0.0001577387156430632, "rewards/count_adherence/mean": 0.8977272510528564, "rewards/count_adherence/std": 0.03214123100042343, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9736841917037964, "rewards/repeat_penalty/std": 0.02813275158405304, "rewards/total_composite/mean": 0.8728748559951782, "rewards/total_composite/std": 0.03264994919300079, "reward": 0.8728748559951782, "reward_std": 0.032649945467710495, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022427460178732872, "sampling/sampling_logp_difference/max": 1.2122278213500977, "sampling/importance_sampling_ratio/min": 0.29753369092941284, "sampling/importance_sampling_ratio/mean": 1.0067881345748901, "sampling/importance_sampling_ratio/max": 1.6595386266708374, "entropy": 0.3090662658214569, "clip_ratio/low_mean": 0.009166883770376444, "clip_ratio/low_min": 0.009166883770376444, "clip_ratio/high_mean": 0.004899043240584433, "clip_ratio/high_max": 0.004899043240584433, "clip_ratio/region_mean": 0.014065927010960877, "reward_total_mean": 0.8728748559951782, "reward_meter_mean": 0.9989354610443115, "reward_meter_std": 0.0001577387156430632, "reward_count_adherence_mean": 0.8977272510528564, "reward_count_adherence_std": 0.03214123100042343, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9736841917037964, "reward_repeat_penalty_std": 0.02813275158405304, "reward_total_composite_mean": 0.8728748559951782, "reward_total_composite_std": 0.03264994919300079} {"timestamp_utc": "2026-04-12T03:04:29Z", "mode": "train", "global_step": 2961, "epoch": 0.11892999156524882, "loss": -0.0069, "grad_norm": 9.514266967773438, "learning_rate": 1.0303030303030304e-06, "num_tokens": 6721217.0, "completions/mean_length": 85.875, "completions/min_length": 83.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.875, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.9182481169700623, "rewards/meter/std": 0.11503338068723679, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9182481169700623, "rewards/total_composite/std": 0.11503338068723679, "reward": 0.9182481169700623, "reward_std": 0.11503338813781738, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06353027373552322, "sampling/sampling_logp_difference/max": 2.5414462089538574, "sampling/importance_sampling_ratio/min": 0.07875242084264755, "sampling/importance_sampling_ratio/mean": 0.9983007907867432, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31395105831325054, "clip_ratio/low_mean": 0.013554217293858528, "clip_ratio/low_min": 0.013554217293858528, "clip_ratio/high_mean": 0.0521149430423975, "clip_ratio/high_max": 0.0521149430423975, "clip_ratio/region_mean": 0.06566916033625603, "reward_total_mean": 0.9182481169700623, "reward_meter_mean": 0.9182481169700623, "reward_meter_std": 0.11503338068723679, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9182481169700623, "reward_total_composite_std": 0.11503338068723679} {"timestamp_utc": "2026-04-12T03:04:34Z", "mode": "train", "global_step": 2962, "epoch": 0.11897015704703377, "loss": -0.0033, "grad_norm": 4.379561424255371, "learning_rate": 1.0272727272727274e-06, "num_tokens": 6722966.0, "completions/mean_length": 63.625, "completions/min_length": 63.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.625, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9992475509643555, "rewards/meter/std": 0.00020958359527867287, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992475509643555, "rewards/total_composite/std": 0.00020958359527867287, "reward": 0.9992475509643555, "reward_std": 0.00020958769891876727, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00583175802603364, "sampling/sampling_logp_difference/max": 1.3357278108596802, "sampling/importance_sampling_ratio/min": 0.26296672224998474, "sampling/importance_sampling_ratio/mean": 1.0014468431472778, "sampling/importance_sampling_ratio/max": 1.7847542762756348, "entropy": 0.009632107801735401, "clip_ratio/low_mean": 0.0019841270986944437, "clip_ratio/low_min": 0.0019841270986944437, "clip_ratio/high_mean": 0.001953125, "clip_ratio/high_max": 0.001953125, "clip_ratio/region_mean": 0.003937252098694444, "reward_total_mean": 0.9992475509643555, "reward_meter_mean": 0.9992475509643555, "reward_meter_std": 0.00020958359527867287, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992475509643555, "reward_total_composite_std": 0.00020958359527867287} {"timestamp_utc": "2026-04-12T03:04:40Z", "mode": "train", "global_step": 2963, "epoch": 0.11901032252881873, "loss": 0.0074, "grad_norm": 2.2698678970336914, "learning_rate": 1.0242424242424242e-06, "num_tokens": 6725671.0, "completions/mean_length": 157.125, "completions/min_length": 152.0, "completions/max_length": 160.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 157.125, "completions/min_terminated_length": 152.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.9989545345306396, "rewards/meter/std": 0.0008401040686294436, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9811090230941772, "rewards/total_composite/std": 0.05032341554760933, "reward": 0.9811090230941772, "reward_std": 0.050323426723480225, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03372993692755699, "sampling/sampling_logp_difference/max": 1.6377344131469727, "sampling/importance_sampling_ratio/min": 0.19442002475261688, "sampling/importance_sampling_ratio/mean": 1.0084627866744995, "sampling/importance_sampling_ratio/max": 1.5869454145431519, "entropy": 0.3180831354111433, "clip_ratio/low_mean": 0.004687500186264515, "clip_ratio/low_min": 0.004687500186264515, "clip_ratio/high_mean": 0.023915290948934853, "clip_ratio/high_max": 0.023915290948934853, "clip_ratio/region_mean": 0.028602791135199368, "reward_total_mean": 0.9811090230941772, "reward_meter_mean": 0.9989545345306396, "reward_meter_std": 0.0008401040686294436, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.9811090230941772, "reward_total_composite_std": 0.05032341554760933} {"timestamp_utc": "2026-04-12T03:04:44Z", "mode": "train", "global_step": 2964, "epoch": 0.11905048801060368, "loss": -0.0008, "grad_norm": 1.2236340045928955, "learning_rate": 1.0212121212121213e-06, "num_tokens": 6727399.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993772506713867, "rewards/meter/std": 6.269344157772139e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993772506713867, "rewards/total_composite/std": 6.269344157772139e-05, "reward": 0.9993772506713867, "reward_std": 6.270247104112059e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.002508246572688222, "sampling/sampling_logp_difference/max": 0.5242347717285156, "sampling/importance_sampling_ratio/min": 0.5920082330703735, "sampling/importance_sampling_ratio/mean": 0.9991680383682251, "sampling/importance_sampling_ratio/max": 1.1811754703521729, "entropy": 0.00610040313040372, "clip_ratio/low_mean": 0.001953125, "clip_ratio/low_min": 0.001953125, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.001953125, "reward_total_mean": 0.9993772506713867, "reward_meter_mean": 0.9993772506713867, "reward_meter_std": 6.269344157772139e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993772506713867, "reward_total_composite_std": 6.269344157772139e-05} {"timestamp_utc": "2026-04-12T03:04:49Z", "mode": "train", "global_step": 2965, "epoch": 0.11909065349238863, "loss": -0.0025, "grad_norm": 0.8599671721458435, "learning_rate": 1.0181818181818183e-06, "num_tokens": 6729286.0, "completions/mean_length": 67.875, "completions/min_length": 67.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.875, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9994280338287354, "rewards/meter/std": 7.756530249025673e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994280338287354, "rewards/total_composite/std": 7.756530249025673e-05, "reward": 0.9994280338287354, "reward_std": 7.756453851470724e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010112082585692406, "sampling/sampling_logp_difference/max": 0.8493509292602539, "sampling/importance_sampling_ratio/min": 0.4276924431324005, "sampling/importance_sampling_ratio/mean": 0.9992300271987915, "sampling/importance_sampling_ratio/max": 1.371290683746338, "entropy": 0.0738078411668539, "clip_ratio/low_mean": 0.005542142200283706, "clip_ratio/low_min": 0.005542142200283706, "clip_ratio/high_mean": 0.0018382353009656072, "clip_ratio/high_max": 0.0018382353009656072, "clip_ratio/region_mean": 0.007380377501249313, "reward_total_mean": 0.9994280338287354, "reward_meter_mean": 0.9994280338287354, "reward_meter_std": 7.756530249025673e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994280338287354, "reward_total_composite_std": 7.756530249025673e-05} {"timestamp_utc": "2026-04-12T03:04:56Z", "mode": "train", "global_step": 2966, "epoch": 0.11913081897417359, "loss": 0.0035, "grad_norm": 1.9423378705978394, "learning_rate": 1.0151515151515152e-06, "num_tokens": 6733705.0, "completions/mean_length": 296.375, "completions/min_length": 291.0, "completions/max_length": 301.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 296.375, "completions/min_terminated_length": 291.0, "completions/max_terminated_length": 301.0, "rewards/meter/mean": 0.9927277565002441, "rewards/meter/std": 0.018433313816785812, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9264706373214722, "rewards/repeat_penalty/std": 0.04159451648592949, "rewards/total_composite/mean": 0.9196349382400513, "rewards/total_composite/std": 0.042633701115846634, "reward": 0.9196349382400513, "reward_std": 0.04263370484113693, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.029081236571073532, "sampling/sampling_logp_difference/max": 1.037057876586914, "sampling/importance_sampling_ratio/min": 0.3544961214065552, "sampling/importance_sampling_ratio/mean": 1.0065155029296875, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3008219301700592, "clip_ratio/low_mean": 0.008422989631071687, "clip_ratio/low_min": 0.008422989631071687, "clip_ratio/high_mean": 0.014326842967420816, "clip_ratio/high_max": 0.014326842967420816, "clip_ratio/region_mean": 0.022749832598492503, "reward_total_mean": 0.9196349382400513, "reward_meter_mean": 0.9927277565002441, "reward_meter_std": 0.018433313816785812, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9264706373214722, "reward_repeat_penalty_std": 0.04159451648592949, "reward_total_composite_mean": 0.9196349382400513, "reward_total_composite_std": 0.042633701115846634} {"timestamp_utc": "2026-04-12T03:05:01Z", "mode": "train", "global_step": 2967, "epoch": 0.11917098445595854, "loss": 0.0063, "grad_norm": 2.049762487411499, "learning_rate": 1.0121212121212122e-06, "num_tokens": 6735749.0, "completions/mean_length": 92.5, "completions/min_length": 91.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.5, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.9977041482925415, "rewards/meter/std": 0.00018006080063059926, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977041482925415, "rewards/total_composite/std": 0.00018006080063059926, "reward": 0.9977041482925415, "reward_std": 0.0001800469763111323, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01588408276438713, "sampling/sampling_logp_difference/max": 1.0536683797836304, "sampling/importance_sampling_ratio/min": 0.34865638613700867, "sampling/importance_sampling_ratio/mean": 1.0007275342941284, "sampling/importance_sampling_ratio/max": 1.9789665937423706, "entropy": 0.11041967011988163, "clip_ratio/low_mean": 0.008021619636565447, "clip_ratio/low_min": 0.008021619636565447, "clip_ratio/high_mean": 0.008167747058905661, "clip_ratio/high_max": 0.008167747058905661, "clip_ratio/region_mean": 0.016189366695471108, "reward_total_mean": 0.9977041482925415, "reward_meter_mean": 0.9977041482925415, "reward_meter_std": 0.00018006080063059926, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9977041482925415, "reward_total_composite_std": 0.00018006080063059926} {"timestamp_utc": "2026-04-12T03:05:10Z", "mode": "train", "global_step": 2968, "epoch": 0.1192111499377435, "loss": 0.0007, "grad_norm": 1.8433130979537964, "learning_rate": 1.0090909090909092e-06, "num_tokens": 6740073.0, "completions/mean_length": 337.5, "completions/min_length": 332.0, "completions/max_length": 347.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 337.5, "completions/min_terminated_length": 332.0, "completions/max_terminated_length": 347.0, "rewards/meter/mean": 0.9978909492492676, "rewards/meter/std": 0.0014596291584894061, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9868420958518982, "rewards/repeat_penalty/std": 0.024363677948713303, "rewards/total_composite/mean": 0.9847573637962341, "rewards/total_composite/std": 0.024200566112995148, "reward": 0.9847573637962341, "reward_std": 0.024200553074479103, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.051781829446554184, "sampling/sampling_logp_difference/max": 2.833930492401123, "sampling/importance_sampling_ratio/min": 0.05878135934472084, "sampling/importance_sampling_ratio/mean": 1.0109304189682007, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4550737701356411, "clip_ratio/low_mean": 0.00747274118475616, "clip_ratio/low_min": 0.00747274118475616, "clip_ratio/high_mean": 0.0354147027246654, "clip_ratio/high_max": 0.0354147027246654, "clip_ratio/region_mean": 0.04288744390942156, "reward_total_mean": 0.9847573637962341, "reward_meter_mean": 0.9978909492492676, "reward_meter_std": 0.0014596291584894061, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9868420958518982, "reward_repeat_penalty_std": 0.024363677948713303, "reward_total_composite_mean": 0.9847573637962341, "reward_total_composite_std": 0.024200566112995148} {"timestamp_utc": "2026-04-12T03:05:14Z", "mode": "train", "global_step": 2969, "epoch": 0.11925131541952846, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.006060606060606e-06, "num_tokens": 6741609.0, "completions/mean_length": 29.0, "completions/min_length": 29.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9957436919212341, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957436919212341, "rewards/total_composite/std": 0.0, "reward": 0.9957436919212341, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0006444337195716798, "sampling/sampling_logp_difference/max": 0.0378272607922554, "sampling/importance_sampling_ratio/min": 0.9628793001174927, "sampling/importance_sampling_ratio/mean": 1.000121831893921, "sampling/importance_sampling_ratio/max": 1.0287812948226929, "entropy": 0.005584679980529472, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9957436919212341, "reward_meter_mean": 0.9957436919212341, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9957436919212341, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:05:19Z", "mode": "train", "global_step": 2970, "epoch": 0.11929148090131342, "loss": 0.0005, "grad_norm": 0.3558986186981201, "learning_rate": 1.0030303030303031e-06, "num_tokens": 6744177.0, "completions/mean_length": 132.0, "completions/min_length": 131.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.0, "completions/min_terminated_length": 131.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.9994352459907532, "rewards/meter/std": 1.4545888916472904e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994352459907532, "rewards/total_composite/std": 1.4545888916472904e-05, "reward": 0.9994352459907532, "reward_std": 1.4537939932779409e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010489360429346561, "sampling/sampling_logp_difference/max": 0.6865215301513672, "sampling/importance_sampling_ratio/min": 0.5033237934112549, "sampling/importance_sampling_ratio/mean": 1.0008372068405151, "sampling/importance_sampling_ratio/max": 1.4119945764541626, "entropy": 0.09148383047431707, "clip_ratio/low_mean": 0.005667578196153045, "clip_ratio/low_min": 0.005667578196153045, "clip_ratio/high_mean": 0.006643245578743517, "clip_ratio/high_max": 0.006643245578743517, "clip_ratio/region_mean": 0.012310823774896562, "reward_total_mean": 0.9994352459907532, "reward_meter_mean": 0.9994352459907532, "reward_meter_std": 1.4545888916472904e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994352459907532, "reward_total_composite_std": 1.4545888916472904e-05} {"timestamp_utc": "2026-04-12T03:05:24Z", "mode": "train", "global_step": 2971, "epoch": 0.11933164638309837, "loss": 0.001, "grad_norm": 0.8809167742729187, "learning_rate": 1.0000000000000002e-06, "num_tokens": 6746402.0, "completions/mean_length": 98.125, "completions/min_length": 98.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.125, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.9994017481803894, "rewards/meter/std": 5.140481152920984e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994017481803894, "rewards/total_composite/std": 5.140481152920984e-05, "reward": 0.9994017481803894, "reward_std": 5.139390123076737e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011422020383179188, "sampling/sampling_logp_difference/max": 1.0734617710113525, "sampling/importance_sampling_ratio/min": 0.5932945609092712, "sampling/importance_sampling_ratio/mean": 1.0018893480300903, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.051710725761950016, "clip_ratio/low_mean": 0.006377551006153226, "clip_ratio/low_min": 0.006377551006153226, "clip_ratio/high_mean": 0.0012755101779475808, "clip_ratio/high_max": 0.0012755101779475808, "clip_ratio/region_mean": 0.007653061184100807, "reward_total_mean": 0.9994017481803894, "reward_meter_mean": 0.9994017481803894, "reward_meter_std": 5.140481152920984e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994017481803894, "reward_total_composite_std": 5.140481152920984e-05} {"timestamp_utc": "2026-04-12T03:05:28Z", "mode": "train", "global_step": 2972, "epoch": 0.11937181186488333, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 9.96969696969697e-07, "num_tokens": 6747826.0, "completions/mean_length": 29.0, "completions/min_length": 29.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9957436919212341, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957436919212341, "rewards/total_composite/std": 0.0, "reward": 0.9957436919212341, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0012666110415011644, "sampling/sampling_logp_difference/max": 0.08357194811105728, "sampling/importance_sampling_ratio/min": 0.9198248982429504, "sampling/importance_sampling_ratio/mean": 1.000320315361023, "sampling/importance_sampling_ratio/max": 1.056694507598877, "entropy": 0.010805643978528678, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9957436919212341, "reward_meter_mean": 0.9957436919212341, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9957436919212341, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:05:35Z", "mode": "train", "global_step": 2973, "epoch": 0.11941197734666828, "loss": 0.0073, "grad_norm": 1.3921781778335571, "learning_rate": 9.93939393939394e-07, "num_tokens": 6751625.0, "completions/mean_length": 248.875, "completions/min_length": 246.0, "completions/max_length": 253.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 248.875, "completions/min_terminated_length": 246.0, "completions/max_terminated_length": 253.0, "rewards/meter/mean": 0.9990034699440002, "rewards/meter/std": 0.00021533701510634273, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9711538553237915, "rewards/repeat_penalty/std": 0.039811473339796066, "rewards/total_composite/mean": 0.9701830148696899, "rewards/total_composite/std": 0.03968827798962593, "reward": 0.9701830148696899, "reward_std": 0.039688270539045334, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020545274019241333, "sampling/sampling_logp_difference/max": 1.2074041366577148, "sampling/importance_sampling_ratio/min": 0.29897233843803406, "sampling/importance_sampling_ratio/mean": 1.0035536289215088, "sampling/importance_sampling_ratio/max": 1.81640625, "entropy": 0.2675408758223057, "clip_ratio/low_mean": 0.002494166372343898, "clip_ratio/low_min": 0.002494166372343898, "clip_ratio/high_mean": 0.012083206791430712, "clip_ratio/high_max": 0.012083206791430712, "clip_ratio/region_mean": 0.01457737316377461, "reward_total_mean": 0.9701830148696899, "reward_meter_mean": 0.9990034699440002, "reward_meter_std": 0.00021533701510634273, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9711538553237915, "reward_repeat_penalty_std": 0.039811473339796066, "reward_total_composite_mean": 0.9701830148696899, "reward_total_composite_std": 0.03968827798962593} {"timestamp_utc": "2026-04-12T03:05:43Z", "mode": "train", "global_step": 2974, "epoch": 0.11945214282845323, "loss": -0.0, "grad_norm": 2.7557997703552246, "learning_rate": 9.90909090909091e-07, "num_tokens": 6755543.0, "completions/mean_length": 292.75, "completions/min_length": 279.0, "completions/max_length": 309.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 292.75, "completions/min_terminated_length": 279.0, "completions/max_terminated_length": 309.0, "rewards/meter/mean": 0.9976239204406738, "rewards/meter/std": 0.0013418762246146798, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.9558823108673096, "rewards/repeat_penalty/std": 0.06852733343839645, "rewards/total_composite/mean": 0.8287225961685181, "rewards/total_composite/std": 0.3412354588508606, "reward": 0.8287225961685181, "reward_std": 0.3412354588508606, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05460209771990776, "sampling/sampling_logp_difference/max": 2.425222635269165, "sampling/importance_sampling_ratio/min": 0.08845842629671097, "sampling/importance_sampling_ratio/mean": 1.0052495002746582, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4930482190102339, "clip_ratio/low_mean": 0.007359597599133849, "clip_ratio/low_min": 0.007359597599133849, "clip_ratio/high_mean": 0.03571874415501952, "clip_ratio/high_max": 0.03571874415501952, "clip_ratio/region_mean": 0.04307834175415337, "reward_total_mean": 0.8287225961685181, "reward_meter_mean": 0.9976239204406738, "reward_meter_std": 0.0013418762246146798, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.9558823108673096, "reward_repeat_penalty_std": 0.06852733343839645, "reward_total_composite_mean": 0.8287225961685181, "reward_total_composite_std": 0.3412354588508606} {"timestamp_utc": "2026-04-12T03:05:48Z", "mode": "train", "global_step": 2975, "epoch": 0.11949230831023819, "loss": -0.0015, "grad_norm": 1.2692075967788696, "learning_rate": 9.87878787878788e-07, "num_tokens": 6757327.0, "completions/mean_length": 68.0, "completions/min_length": 68.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.0, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9994783401489258, "rewards/meter/std": 9.867444896372035e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994783401489258, "rewards/total_composite/std": 9.867444896372035e-05, "reward": 0.9994783401489258, "reward_std": 9.867054177448153e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009787042625248432, "sampling/sampling_logp_difference/max": 0.6712450981140137, "sampling/importance_sampling_ratio/min": 0.5110718607902527, "sampling/importance_sampling_ratio/mean": 0.9991775751113892, "sampling/importance_sampling_ratio/max": 1.392788290977478, "entropy": 0.07629173761233687, "clip_ratio/low_mean": 0.0018382353009656072, "clip_ratio/low_min": 0.0018382353009656072, "clip_ratio/high_mean": 0.009191176504828036, "clip_ratio/high_max": 0.009191176504828036, "clip_ratio/region_mean": 0.011029411805793643, "reward_total_mean": 0.9994783401489258, "reward_meter_mean": 0.9994783401489258, "reward_meter_std": 9.867444896372035e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994783401489258, "reward_total_composite_std": 9.867444896372035e-05} {"timestamp_utc": "2026-04-12T03:05:54Z", "mode": "train", "global_step": 2976, "epoch": 0.11953247379202314, "loss": -0.0012, "grad_norm": 1.1907378435134888, "learning_rate": 9.84848484848485e-07, "num_tokens": 6759949.0, "completions/mean_length": 131.75, "completions/min_length": 130.0, "completions/max_length": 132.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.75, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.9993938207626343, "rewards/meter/std": 7.404623465845361e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993938207626343, "rewards/total_composite/std": 7.404623465845361e-05, "reward": 0.9993938207626343, "reward_std": 7.40409450372681e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008480089716613293, "sampling/sampling_logp_difference/max": 0.5749435424804688, "sampling/importance_sampling_ratio/min": 0.5627366304397583, "sampling/importance_sampling_ratio/mean": 1.0010247230529785, "sampling/importance_sampling_ratio/max": 1.4190884828567505, "entropy": 0.07698143133893609, "clip_ratio/low_mean": 0.000961538462433964, "clip_ratio/low_min": 0.000961538462433964, "clip_ratio/high_mean": 0.006628788076341152, "clip_ratio/high_max": 0.006628788076341152, "clip_ratio/region_mean": 0.007590326538775116, "reward_total_mean": 0.9993938207626343, "reward_meter_mean": 0.9993938207626343, "reward_meter_std": 7.404623465845361e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993938207626343, "reward_total_composite_std": 7.404623465845361e-05} {"timestamp_utc": "2026-04-12T03:05:59Z", "mode": "train", "global_step": 2977, "epoch": 0.1195726392738081, "loss": 0.0003, "grad_norm": 0.025195496156811714, "learning_rate": 9.818181818181818e-07, "num_tokens": 6762117.0, "completions/mean_length": 98.0, "completions/min_length": 98.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.0, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9994038343429565, "rewards/meter/std": 3.7851127672183793e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994038343429565, "rewards/total_composite/std": 3.7851127672183793e-06, "reward": 0.9994038343429565, "reward_std": 3.7985312246746616e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0029753749258816242, "sampling/sampling_logp_difference/max": 0.2579946517944336, "sampling/importance_sampling_ratio/min": 0.8075622916221619, "sampling/importance_sampling_ratio/mean": 1.000174880027771, "sampling/importance_sampling_ratio/max": 1.2943319082260132, "entropy": 0.0308038501534611, "clip_ratio/low_mean": 0.0025510203558951616, "clip_ratio/low_min": 0.0025510203558951616, "clip_ratio/high_mean": 0.0012755101779475808, "clip_ratio/high_max": 0.0012755101779475808, "clip_ratio/region_mean": 0.0038265305338427424, "reward_total_mean": 0.9994038343429565, "reward_meter_mean": 0.9994038343429565, "reward_meter_std": 3.7851127672183793e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994038343429565, "reward_total_composite_std": 3.7851127672183793e-06} {"timestamp_utc": "2026-04-12T03:06:05Z", "mode": "train", "global_step": 2978, "epoch": 0.11961280475559305, "loss": -0.0016, "grad_norm": 3.375030994415283, "learning_rate": 9.787878787878788e-07, "num_tokens": 6764878.0, "completions/mean_length": 167.125, "completions/min_length": 162.0, "completions/max_length": 173.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 167.125, "completions/min_terminated_length": 162.0, "completions/max_terminated_length": 173.0, "rewards/meter/mean": 0.9967660903930664, "rewards/meter/std": 0.00537865748628974, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.9690313339233398, "rewards/total_composite/std": 0.05059126019477844, "reward": 0.9690313339233398, "reward_std": 0.050591256469488144, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03752928227186203, "sampling/sampling_logp_difference/max": 1.3720521926879883, "sampling/importance_sampling_ratio/min": 0.2535859942436218, "sampling/importance_sampling_ratio/mean": 1.0075362920761108, "sampling/importance_sampling_ratio/max": 1.8077061176300049, "entropy": 0.30791075527668, "clip_ratio/low_mean": 0.003756533144041896, "clip_ratio/low_min": 0.003756533144041896, "clip_ratio/high_mean": 0.02096639317460358, "clip_ratio/high_max": 0.02096639317460358, "clip_ratio/region_mean": 0.024722926318645477, "reward_total_mean": 0.9690313339233398, "reward_meter_mean": 0.9967660903930664, "reward_meter_std": 0.00537865748628974, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_total_composite_mean": 0.9690313339233398, "reward_total_composite_std": 0.05059126019477844} {"timestamp_utc": "2026-04-12T03:06:09Z", "mode": "train", "global_step": 2979, "epoch": 0.119652970237378, "loss": 0.015, "grad_norm": 6.203634738922119, "learning_rate": 9.757575757575759e-07, "num_tokens": 6766789.0, "completions/mean_length": 70.875, "completions/min_length": 70.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.875, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9866373538970947, "rewards/meter/std": 0.01733180694282055, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9866373538970947, "rewards/total_composite/std": 0.01733180694282055, "reward": 0.9866373538970947, "reward_std": 0.017331810668110847, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025828473269939423, "sampling/sampling_logp_difference/max": 0.6619188785552979, "sampling/importance_sampling_ratio/min": 0.5158604979515076, "sampling/importance_sampling_ratio/mean": 1.0079153776168823, "sampling/importance_sampling_ratio/max": 1.5757840871810913, "entropy": 0.27640925347805023, "clip_ratio/low_mean": 0.008778364514000714, "clip_ratio/low_min": 0.008778364514000714, "clip_ratio/high_mean": 0.02296627010218799, "clip_ratio/high_max": 0.02296627010218799, "clip_ratio/region_mean": 0.031744634616188705, "reward_total_mean": 0.9866373538970947, "reward_meter_mean": 0.9866373538970947, "reward_meter_std": 0.01733180694282055, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9866373538970947, "reward_total_composite_std": 0.01733180694282055} {"timestamp_utc": "2026-04-12T03:06:14Z", "mode": "train", "global_step": 2980, "epoch": 0.11969313571916296, "loss": 0.0156, "grad_norm": 5.659327507019043, "learning_rate": 9.72727272727273e-07, "num_tokens": 6768713.0, "completions/mean_length": 68.5, "completions/min_length": 68.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.5, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9975705742835999, "rewards/meter/std": 0.004985023755580187, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9975705742835999, "rewards/total_composite/std": 0.004985023755580187, "reward": 0.9975705742835999, "reward_std": 0.004985013976693153, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016337888315320015, "sampling/sampling_logp_difference/max": 1.1524267196655273, "sampling/importance_sampling_ratio/min": 0.3158693015575409, "sampling/importance_sampling_ratio/mean": 1.0029165744781494, "sampling/importance_sampling_ratio/max": 1.6995705366134644, "entropy": 0.11373988864943385, "clip_ratio/low_mean": 0.0053571430034935474, "clip_ratio/low_min": 0.0053571430034935474, "clip_ratio/high_mean": 0.007247899193316698, "clip_ratio/high_max": 0.007247899193316698, "clip_ratio/region_mean": 0.012605042196810246, "reward_total_mean": 0.9975705742835999, "reward_meter_mean": 0.9975705742835999, "reward_meter_std": 0.004985023755580187, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9975705742835999, "reward_total_composite_std": 0.004985023755580187} {"timestamp_utc": "2026-04-12T03:06:18Z", "mode": "train", "global_step": 2981, "epoch": 0.11973330120094791, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 9.696969696969698e-07, "num_tokens": 6770249.0, "completions/mean_length": 28.0, "completions/min_length": 28.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.0, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9943599104881287, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9943599104881287, "rewards/total_composite/std": 0.0, "reward": 0.9943599104881287, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0001402015914209187, "sampling/sampling_logp_difference/max": 0.0023617574479430914, "sampling/importance_sampling_ratio/min": 0.9999930262565613, "sampling/importance_sampling_ratio/mean": 1.0001403093338013, "sampling/importance_sampling_ratio/max": 1.0023645162582397, "entropy": 0.0012292767642065883, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9943599104881287, "reward_meter_mean": 0.9943599104881287, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9943599104881287, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:06:24Z", "mode": "train", "global_step": 2982, "epoch": 0.11977346668273287, "loss": -0.0015, "grad_norm": 2.0111052989959717, "learning_rate": 9.666666666666668e-07, "num_tokens": 6772571.0, "completions/mean_length": 118.25, "completions/min_length": 117.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.25, "completions/min_terminated_length": 117.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.9989663362503052, "rewards/meter/std": 0.0008744557853788137, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989663362503052, "rewards/total_composite/std": 0.0008744557853788137, "reward": 0.9989663362503052, "reward_std": 0.0008744720835238695, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03213733434677124, "sampling/sampling_logp_difference/max": 1.6665115356445312, "sampling/importance_sampling_ratio/min": 0.18890489637851715, "sampling/importance_sampling_ratio/mean": 1.0011169910430908, "sampling/importance_sampling_ratio/max": 1.6112860441207886, "entropy": 0.2766532879322767, "clip_ratio/low_mean": 0.0032051282469183207, "clip_ratio/low_min": 0.0032051282469183207, "clip_ratio/high_mean": 0.02634434588253498, "clip_ratio/high_max": 0.02634434588253498, "clip_ratio/region_mean": 0.0295494741294533, "reward_total_mean": 0.9989663362503052, "reward_meter_mean": 0.9989663362503052, "reward_meter_std": 0.0008744557853788137, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9989663362503052, "reward_total_composite_std": 0.0008744557853788137} {"timestamp_utc": "2026-04-12T03:06:29Z", "mode": "train", "global_step": 2983, "epoch": 0.11981363216451782, "loss": 0.0056, "grad_norm": 4.000948905944824, "learning_rate": 9.636363636363636e-07, "num_tokens": 6774679.0, "completions/mean_length": 100.5, "completions/min_length": 99.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.5, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.9993005990982056, "rewards/meter/std": 0.00027800656971521676, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9249999523162842, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.9243372678756714, "rewards/total_composite/std": 0.1032847911119461, "reward": 0.9243372678756714, "reward_std": 0.10328476876020432, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023382917046546936, "sampling/sampling_logp_difference/max": 1.0612468719482422, "sampling/importance_sampling_ratio/min": 0.3460240960121155, "sampling/importance_sampling_ratio/mean": 1.0000145435333252, "sampling/importance_sampling_ratio/max": 1.835827350616455, "entropy": 0.17713584937155247, "clip_ratio/low_mean": 0.0061881187139078975, "clip_ratio/low_min": 0.0061881187139078975, "clip_ratio/high_mean": 0.016288378508761525, "clip_ratio/high_max": 0.016288378508761525, "clip_ratio/region_mean": 0.022476497222669423, "reward_total_mean": 0.9243372678756714, "reward_meter_mean": 0.9993005990982056, "reward_meter_std": 0.00027800656971521676, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9249999523162842, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.9243372678756714, "reward_total_composite_std": 0.1032847911119461} {"timestamp_utc": "2026-04-12T03:06:33Z", "mode": "train", "global_step": 2984, "epoch": 0.11985379764630277, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 9.606060606060607e-07, "num_tokens": 6776207.0, "completions/mean_length": 35.0, "completions/min_length": 35.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9980231523513794, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980231523513794, "rewards/total_composite/std": 0.0, "reward": 0.9980231523513794, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002929195179603994, "sampling/sampling_logp_difference/max": 0.00473133847117424, "sampling/importance_sampling_ratio/min": 0.995279848575592, "sampling/importance_sampling_ratio/mean": 1.0001977682113647, "sampling/importance_sampling_ratio/max": 1.004080891609192, "entropy": 0.003427958843531087, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9980231523513794, "reward_meter_mean": 0.9980231523513794, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9980231523513794, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:06:39Z", "mode": "train", "global_step": 2985, "epoch": 0.11989396312808773, "loss": 0.0071, "grad_norm": 2.747091770172119, "learning_rate": 9.575757575757577e-07, "num_tokens": 6779042.0, "completions/mean_length": 167.375, "completions/min_length": 164.0, "completions/max_length": 171.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 167.375, "completions/min_terminated_length": 164.0, "completions/max_terminated_length": 171.0, "rewards/meter/mean": 0.9969633221626282, "rewards/meter/std": 0.003777019679546356, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9305555820465088, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.9275848865509033, "rewards/total_composite/std": 0.05462638661265373, "reward": 0.9275848865509033, "reward_std": 0.05462638661265373, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03595529869198799, "sampling/sampling_logp_difference/max": 1.5971007347106934, "sampling/importance_sampling_ratio/min": 0.20248271524906158, "sampling/importance_sampling_ratio/mean": 1.004103422164917, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2825808133929968, "clip_ratio/low_mean": 0.018017811933532357, "clip_ratio/low_min": 0.018017811933532357, "clip_ratio/high_mean": 0.011204644571989775, "clip_ratio/high_max": 0.011204644571989775, "clip_ratio/region_mean": 0.029222456505522132, "reward_total_mean": 0.9275848865509033, "reward_meter_mean": 0.9969633221626282, "reward_meter_std": 0.003777019679546356, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9305555820465088, "reward_repeat_penalty_std": 0.05750546231865883, "reward_total_composite_mean": 0.9275848865509033, "reward_total_composite_std": 0.05462638661265373} {"timestamp_utc": "2026-04-12T03:06:45Z", "mode": "train", "global_step": 2986, "epoch": 0.11993412860987268, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 9.545454545454548e-07, "num_tokens": 6780498.0, "completions/mean_length": 32.0, "completions/min_length": 32.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "reward": 0.9992982149124146, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 5.2063212933717296e-05, "sampling/sampling_logp_difference/max": 0.0009029797511175275, "sampling/importance_sampling_ratio/min": 0.9995164275169373, "sampling/importance_sampling_ratio/mean": 1.0000447034835815, "sampling/importance_sampling_ratio/max": 1.0009034872055054, "entropy": 0.00046791824570391327, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9992982149124146, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:06:50Z", "mode": "train", "global_step": 2987, "epoch": 0.11997429409165764, "loss": -0.004, "grad_norm": 2.3871347904205322, "learning_rate": 9.515151515151516e-07, "num_tokens": 6782659.0, "completions/mean_length": 71.125, "completions/min_length": 70.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.125, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9993984699249268, "rewards/meter/std": 9.4733273726888e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993984699249268, "rewards/total_composite/std": 9.4733273726888e-05, "reward": 0.9993984699249268, "reward_std": 9.472780220676214e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010385324247181416, "sampling/sampling_logp_difference/max": 0.6226983070373535, "sampling/importance_sampling_ratio/min": 0.5364948511123657, "sampling/importance_sampling_ratio/mean": 1.0039633512496948, "sampling/importance_sampling_ratio/max": 1.32568359375, "entropy": 0.08504062425345182, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0034966744715347886, "clip_ratio/high_max": 0.0034966744715347886, "clip_ratio/region_mean": 0.0034966744715347886, "reward_total_mean": 0.9993984699249268, "reward_meter_mean": 0.9993984699249268, "reward_meter_std": 9.4733273726888e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993984699249268, "reward_total_composite_std": 9.4733273726888e-05} {"timestamp_utc": "2026-04-12T03:06:55Z", "mode": "train", "global_step": 2988, "epoch": 0.12001445957344259, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 9.484848484848485e-07, "num_tokens": 6784427.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00022603126126341522, "sampling/sampling_logp_difference/max": 0.003381013870239258, "sampling/importance_sampling_ratio/min": 0.9995951652526855, "sampling/importance_sampling_ratio/mean": 1.0002238750457764, "sampling/importance_sampling_ratio/max": 1.0033867359161377, "entropy": 0.0022001640463713557, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:07:01Z", "mode": "train", "global_step": 2989, "epoch": 0.12005462505522754, "loss": 0.0025, "grad_norm": 3.191429615020752, "learning_rate": 9.454545454545455e-07, "num_tokens": 6786878.0, "completions/mean_length": 133.375, "completions/min_length": 128.0, "completions/max_length": 137.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 133.375, "completions/min_terminated_length": 128.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.9979535937309265, "rewards/meter/std": 0.0034614717587828636, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.9622775316238403, "rewards/total_composite/std": 0.06554586440324783, "reward": 0.9622775316238403, "reward_std": 0.06554586440324783, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03642260655760765, "sampling/sampling_logp_difference/max": 2.4946718215942383, "sampling/importance_sampling_ratio/min": 0.08252352476119995, "sampling/importance_sampling_ratio/mean": 0.9995792508125305, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.24928219243884087, "clip_ratio/low_mean": 0.0028013800038024783, "clip_ratio/low_min": 0.0028013800038024783, "clip_ratio/high_mean": 0.02337344060651958, "clip_ratio/high_max": 0.02337344060651958, "clip_ratio/region_mean": 0.026174820610322058, "reward_total_mean": 0.9622775316238403, "reward_meter_mean": 0.9979535937309265, "reward_meter_std": 0.0034614717587828636, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.9622775316238403, "reward_total_composite_std": 0.06554586440324783} {"timestamp_utc": "2026-04-12T03:07:06Z", "mode": "train", "global_step": 2990, "epoch": 0.1200947905370125, "loss": -0.0015, "grad_norm": 3.7938873767852783, "learning_rate": 9.424242424242425e-07, "num_tokens": 6789021.0, "completions/mean_length": 92.875, "completions/min_length": 91.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.875, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.997610330581665, "rewards/meter/std": 0.00025703865685500205, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.997610330581665, "rewards/total_composite/std": 0.00025703865685500205, "reward": 0.997610330581665, "reward_std": 0.0002570512588135898, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01364238653331995, "sampling/sampling_logp_difference/max": 1.0927095413208008, "sampling/importance_sampling_ratio/min": 0.33530673384666443, "sampling/importance_sampling_ratio/mean": 0.9976533055305481, "sampling/importance_sampling_ratio/max": 1.416971206665039, "entropy": 0.09223943296819925, "clip_ratio/low_mean": 0.004032257944345474, "clip_ratio/low_min": 0.004032257944345474, "clip_ratio/high_mean": 0.008035918464884162, "clip_ratio/high_max": 0.008035918464884162, "clip_ratio/region_mean": 0.012068176409229636, "reward_total_mean": 0.997610330581665, "reward_meter_mean": 0.997610330581665, "reward_meter_std": 0.00025703865685500205, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.997610330581665, "reward_total_composite_std": 0.00025703865685500205} {"timestamp_utc": "2026-04-12T03:07:11Z", "mode": "train", "global_step": 2991, "epoch": 0.12013495601879745, "loss": -0.0063, "grad_norm": 2.051816463470459, "learning_rate": 9.393939393939395e-07, "num_tokens": 6790862.0, "completions/mean_length": 66.125, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.125, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9928753972053528, "rewards/meter/std": 0.0007522930391132832, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9928753972053528, "rewards/total_composite/std": 0.0007522930391132832, "reward": 0.9928753972053528, "reward_std": 0.000752286403439939, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024633267894387245, "sampling/sampling_logp_difference/max": 1.791666030883789, "sampling/importance_sampling_ratio/min": 0.16668224334716797, "sampling/importance_sampling_ratio/mean": 1.0033050775527954, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.14550988655537367, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/high_mean": 0.0056535504991188645, "clip_ratio/high_max": 0.0056535504991188645, "clip_ratio/region_mean": 0.007547489949502051, "reward_total_mean": 0.9928753972053528, "reward_meter_mean": 0.9928753972053528, "reward_meter_std": 0.0007522930391132832, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9928753972053528, "reward_total_composite_std": 0.0007522930391132832} {"timestamp_utc": "2026-04-12T03:07:16Z", "mode": "train", "global_step": 2992, "epoch": 0.1201751215005824, "loss": 0.0024, "grad_norm": 3.857464075088501, "learning_rate": 9.363636363636365e-07, "num_tokens": 6792639.0, "completions/mean_length": 61.125, "completions/min_length": 61.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.125, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9971661567687988, "rewards/meter/std": 0.00047978354268707335, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9971661567687988, "rewards/total_composite/std": 0.00047978354268707335, "reward": 0.9971661567687988, "reward_std": 0.0004797954170498997, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00791865587234497, "sampling/sampling_logp_difference/max": 1.693861484527588, "sampling/importance_sampling_ratio/min": 0.18380838632583618, "sampling/importance_sampling_ratio/mean": 0.9966216087341309, "sampling/importance_sampling_ratio/max": 1.1078962087631226, "entropy": 0.025716731324791908, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.006147540640085936, "clip_ratio/high_max": 0.006147540640085936, "clip_ratio/region_mean": 0.006147540640085936, "reward_total_mean": 0.9971661567687988, "reward_meter_mean": 0.9971661567687988, "reward_meter_std": 0.00047978354268707335, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9971661567687988, "reward_total_composite_std": 0.00047978354268707335} {"timestamp_utc": "2026-04-12T03:07:24Z", "mode": "train", "global_step": 2993, "epoch": 0.12021528698236736, "loss": 0.0031, "grad_norm": 1.920817255973816, "learning_rate": 9.333333333333334e-07, "num_tokens": 6797268.0, "completions/mean_length": 328.625, "completions/min_length": 325.0, "completions/max_length": 332.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 328.625, "completions/min_terminated_length": 325.0, "completions/max_terminated_length": 332.0, "rewards/meter/mean": 0.9988096952438354, "rewards/meter/std": 0.0004608924500644207, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9144736528396606, "rewards/repeat_penalty/std": 0.05582422763109207, "rewards/total_composite/mean": 0.9133861064910889, "rewards/total_composite/std": 0.05579417571425438, "reward": 0.9133861064910889, "reward_std": 0.055794164538383484, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03226817771792412, "sampling/sampling_logp_difference/max": 1.3698234558105469, "sampling/importance_sampling_ratio/min": 0.2541518211364746, "sampling/importance_sampling_ratio/mean": 1.007267951965332, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3455524481832981, "clip_ratio/low_mean": 0.011799397761933506, "clip_ratio/low_min": 0.011799397761933506, "clip_ratio/high_mean": 0.014776158845052123, "clip_ratio/high_max": 0.014776158845052123, "clip_ratio/region_mean": 0.02657555660698563, "reward_total_mean": 0.9133861064910889, "reward_meter_mean": 0.9988096952438354, "reward_meter_std": 0.0004608924500644207, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9144736528396606, "reward_repeat_penalty_std": 0.05582422763109207, "reward_total_composite_mean": 0.9133861064910889, "reward_total_composite_std": 0.05579417571425438} {"timestamp_utc": "2026-04-12T03:07:29Z", "mode": "train", "global_step": 2994, "epoch": 0.12025545246415231, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 9.303030303030304e-07, "num_tokens": 6798980.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00023179415438789874, "sampling/sampling_logp_difference/max": 0.0038368557579815388, "sampling/importance_sampling_ratio/min": 0.99946129322052, "sampling/importance_sampling_ratio/mean": 1.0002259016036987, "sampling/importance_sampling_ratio/max": 1.0038442611694336, "entropy": 0.0016921081114560366, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:07:34Z", "mode": "train", "global_step": 2995, "epoch": 0.12029561794593727, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 9.272727272727273e-07, "num_tokens": 6800420.0, "completions/mean_length": 29.0, "completions/min_length": 29.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9957436919212341, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957436919212341, "rewards/total_composite/std": 0.0, "reward": 0.9957436919212341, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00022380216978490353, "sampling/sampling_logp_difference/max": 0.008112169802188873, "sampling/importance_sampling_ratio/min": 0.9919206500053406, "sampling/importance_sampling_ratio/mean": 1.0001194477081299, "sampling/importance_sampling_ratio/max": 1.0056308507919312, "entropy": 0.00262832005682867, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9957436919212341, "reward_meter_mean": 0.9957436919212341, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9957436919212341, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:07:40Z", "mode": "train", "global_step": 2996, "epoch": 0.12033578342772222, "loss": -0.0, "grad_norm": 0.4544410705566406, "learning_rate": 9.242424242424244e-07, "num_tokens": 6802843.0, "completions/mean_length": 131.875, "completions/min_length": 130.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.875, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.9994057416915894, "rewards/meter/std": 4.619282117346302e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994057416915894, "rewards/total_composite/std": 4.619282117346302e-05, "reward": 0.9994057416915894, "reward_std": 4.6202192606870085e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007839047349989414, "sampling/sampling_logp_difference/max": 1.0239849090576172, "sampling/importance_sampling_ratio/min": 0.3591609001159668, "sampling/importance_sampling_ratio/mean": 1.0020174980163574, "sampling/importance_sampling_ratio/max": 1.366064429283142, "entropy": 0.0709062721580267, "clip_ratio/low_mean": 0.000961538462433964, "clip_ratio/low_min": 0.000961538462433964, "clip_ratio/high_mean": 0.0009469697251915932, "clip_ratio/high_max": 0.0009469697251915932, "clip_ratio/region_mean": 0.0019085081876255572, "reward_total_mean": 0.9994057416915894, "reward_meter_mean": 0.9994057416915894, "reward_meter_std": 4.619282117346302e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994057416915894, "reward_total_composite_std": 4.619282117346302e-05} {"timestamp_utc": "2026-04-12T03:07:50Z", "mode": "train", "global_step": 2997, "epoch": 0.12037594890950717, "loss": -0.0832, "grad_norm": 2.213428020477295, "learning_rate": 9.212121212121213e-07, "num_tokens": 6806751.0, "completions/mean_length": 507.5, "completions/min_length": 497.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.5, "completions/mean_terminated_length": 503.0, "completions/min_terminated_length": 497.0, "completions/max_terminated_length": 512.0, "rewards/meter/mean": 0.9938823580741882, "rewards/meter/std": 0.00720559898763895, "rewards/count_adherence/mean": 0.8472222089767456, "rewards/count_adherence/std": 0.0257172379642725, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9873470664024353, "rewards/repeat_penalty/std": 0.01747622899711132, "rewards/total_composite/mean": 0.8312985897064209, "rewards/total_composite/std": 0.026868145912885666, "reward": 0.8312985897064209, "reward_std": 0.026868147775530815, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06921987235546112, "sampling/sampling_logp_difference/max": 2.2429420948028564, "sampling/importance_sampling_ratio/min": 0.10614575445652008, "sampling/importance_sampling_ratio/mean": 1.0148636102676392, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.33224867284297943, "clip_ratio/low_mean": 0.010527235455811024, "clip_ratio/low_min": 0.010527235455811024, "clip_ratio/high_mean": 0.012055294588208199, "clip_ratio/high_max": 0.012055294588208199, "clip_ratio/region_mean": 0.022582530044019222, "reward_total_mean": 0.8312985897064209, "reward_meter_mean": 0.9938823580741882, "reward_meter_std": 0.00720559898763895, "reward_count_adherence_mean": 0.8472222089767456, "reward_count_adherence_std": 0.0257172379642725, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9873470664024353, "reward_repeat_penalty_std": 0.01747622899711132, "reward_total_composite_mean": 0.8312985897064209, "reward_total_composite_std": 0.026868145912885666} {"timestamp_utc": "2026-04-12T03:07:55Z", "mode": "train", "global_step": 2998, "epoch": 0.12041611439129213, "loss": -0.0063, "grad_norm": 5.6068010330200195, "learning_rate": 9.181818181818182e-07, "num_tokens": 6808791.0, "completions/mean_length": 86.0, "completions/min_length": 84.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.0, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.9505056738853455, "rewards/meter/std": 0.01044482085853815, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9505056738853455, "rewards/total_composite/std": 0.01044482085853815, "reward": 0.9505056738853455, "reward_std": 0.010444814339280128, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.046680573374032974, "sampling/sampling_logp_difference/max": 1.9859023094177246, "sampling/importance_sampling_ratio/min": 0.1372566968202591, "sampling/importance_sampling_ratio/mean": 1.004783272743225, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23842502385377884, "clip_ratio/low_mean": 0.010312846396118402, "clip_ratio/low_min": 0.010312846396118402, "clip_ratio/high_mean": 0.03046388761140406, "clip_ratio/high_max": 0.03046388761140406, "clip_ratio/region_mean": 0.040776734007522464, "reward_total_mean": 0.9505056738853455, "reward_meter_mean": 0.9505056738853455, "reward_meter_std": 0.01044482085853815, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9505056738853455, "reward_total_composite_std": 0.01044482085853815} {"timestamp_utc": "2026-04-12T03:08:00Z", "mode": "train", "global_step": 2999, "epoch": 0.12045627987307708, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 9.151515151515153e-07, "num_tokens": 6810823.0, "completions/mean_length": 80.0, "completions/min_length": 80.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.0, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "reward": 0.7190229296684265, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0003676189517136663, "sampling/sampling_logp_difference/max": 0.009273192845284939, "sampling/importance_sampling_ratio/min": 0.9973099231719971, "sampling/importance_sampling_ratio/mean": 1.0003594160079956, "sampling/importance_sampling_ratio/max": 1.0093164443969727, "entropy": 0.0032808549876790494, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7190229296684265, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:08:05Z", "mode": "train", "global_step": 3000, "epoch": 0.12049644535486204, "loss": -0.003, "grad_norm": 2.610877513885498, "learning_rate": 9.121212121212122e-07, "num_tokens": 6812551.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.7007421255111694, "rewards/meter/std": 0.24578078091144562, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7007421255111694, "rewards/total_composite/std": 0.24578078091144562, "reward": 0.7007421255111694, "reward_std": 0.2457807958126068, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0048951394855976105, "sampling/sampling_logp_difference/max": 1.485097885131836, "sampling/importance_sampling_ratio/min": 0.22648018598556519, "sampling/importance_sampling_ratio/mean": 0.9997680187225342, "sampling/importance_sampling_ratio/max": 1.2230128049850464, "entropy": 0.01205012173159048, "clip_ratio/low_mean": 0.002314814832061529, "clip_ratio/low_min": 0.002314814832061529, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.002314814832061529, "reward_total_mean": 0.7007421255111694, "reward_meter_mean": 0.7007421255111694, "reward_meter_std": 0.24578078091144562, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7007421255111694, "reward_total_composite_std": 0.24578078091144562} {"timestamp_utc": "2026-04-12T03:09:24Z", "mode": "eval", "global_step": 3000, "epoch": 0.12049644535486204, "eval_loss": NaN, "eval_runtime": 79.4116, "eval_samples_per_second": 1.31, "eval_steps_per_second": 0.164, "eval_num_tokens": 6812551.0, "eval_completions/mean_length": 213.93269230769232, "eval_completions/min_length": 61.23076923076923, "eval_completions/max_length": 421.53846153846155, "eval_completions/clipped_ratio": 0.009615384615384616, "eval_completions/mean_terminated_length": 210.74725341796875, "eval_completions/min_terminated_length": 61.23076923076923, "eval_completions/max_terminated_length": 412.7692307692308, "eval_rewards/meter/mean": 0.7926333088141221, "eval_rewards/meter/std": 0.3242100785629681, "eval_rewards/count_adherence/mean": 0.9582245624982394, "eval_rewards/count_adherence/std": 0.06270856238328494, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/repeat_penalty/mean": 0.9439093745671786, "eval_rewards/repeat_penalty/std": 0.07702494555940995, "eval_rewards/total_composite/mean": 0.7239083968676053, "eval_rewards/total_composite/std": 0.3284210069821431, "eval_reward": 0.7239083968676053, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.03129683048106157, "eval_sampling/sampling_logp_difference/max": 1.215673538354727, "eval_sampling/importance_sampling_ratio/min": 0.3013949474463096, "eval_sampling/importance_sampling_ratio/mean": 1.0090940640522883, "eval_sampling/importance_sampling_ratio/max": 1.577669803912823, "eval_entropy": 0.3522038482702695, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7239083968676053, "eval_reward_meter_mean": 0.7926333088141221, "eval_reward_meter_std": 0.3242100785629681, "eval_reward_count_adherence_mean": 0.9582245624982394, "eval_reward_count_adherence_std": 0.06270856238328494, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_repeat_penalty_mean": 0.9439093745671786, "eval_reward_repeat_penalty_std": 0.07702494555940995, "eval_reward_total_composite_mean": 0.7239083968676053, "eval_reward_total_composite_std": 0.3284210069821431} {"timestamp_utc": "2026-04-12T03:09:32Z", "mode": "train", "global_step": 3001, "epoch": 0.12053661083664699, "loss": -0.0098, "grad_norm": 12.373333930969238, "learning_rate": 9.090909090909091e-07, "num_tokens": 6814326.0, "completions/mean_length": 62.875, "completions/min_length": 60.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.875, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.8256336450576782, "rewards/meter/std": 0.3046148419380188, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8256336450576782, "rewards/total_composite/std": 0.3046148419380188, "reward": 0.8256336450576782, "reward_std": 0.3046148419380188, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06896661221981049, "sampling/sampling_logp_difference/max": 1.8012008666992188, "sampling/importance_sampling_ratio/min": 0.16510051488876343, "sampling/importance_sampling_ratio/mean": 1.0055382251739502, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.28370110131800175, "clip_ratio/low_mean": 0.014480874873697758, "clip_ratio/low_min": 0.014480874873697758, "clip_ratio/high_mean": 0.03723430214449763, "clip_ratio/high_max": 0.03723430214449763, "clip_ratio/region_mean": 0.05171517701819539, "reward_total_mean": 0.8256336450576782, "reward_meter_mean": 0.8256336450576782, "reward_meter_std": 0.3046148419380188, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8256336450576782, "reward_total_composite_std": 0.3046148419380188} {"timestamp_utc": "2026-04-12T03:09:37Z", "mode": "train", "global_step": 3002, "epoch": 0.12057677631843194, "loss": 0.0072, "grad_norm": 3.2883293628692627, "learning_rate": 9.060606060606062e-07, "num_tokens": 6816267.0, "completions/mean_length": 78.625, "completions/min_length": 77.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.625, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9988808631896973, "rewards/meter/std": 0.00034191427403129637, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9988808631896973, "rewards/total_composite/std": 0.00034191427403129637, "reward": 0.9988808631896973, "reward_std": 0.00034191476879641414, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02113337628543377, "sampling/sampling_logp_difference/max": 1.1642122268676758, "sampling/importance_sampling_ratio/min": 0.3121684789657593, "sampling/importance_sampling_ratio/mean": 1.0053126811981201, "sampling/importance_sampling_ratio/max": 1.6066337823867798, "entropy": 0.18326533026993275, "clip_ratio/low_mean": 0.0031650641467422247, "clip_ratio/low_min": 0.0031650641467422247, "clip_ratio/high_mean": 0.01416189968585968, "clip_ratio/high_max": 0.01416189968585968, "clip_ratio/region_mean": 0.017326963832601905, "reward_total_mean": 0.9988808631896973, "reward_meter_mean": 0.9988808631896973, "reward_meter_std": 0.00034191427403129637, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9988808631896973, "reward_total_composite_std": 0.00034191427403129637} {"timestamp_utc": "2026-04-12T03:09:43Z", "mode": "train", "global_step": 3003, "epoch": 0.1206169418002169, "loss": 0.0131, "grad_norm": 3.6394121646881104, "learning_rate": 9.030303030303031e-07, "num_tokens": 6818772.0, "completions/mean_length": 137.125, "completions/min_length": 132.0, "completions/max_length": 142.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 137.125, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 142.0, "rewards/meter/mean": 0.9025046229362488, "rewards/meter/std": 0.2555387020111084, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.8669579029083252, "rewards/total_composite/std": 0.24918241798877716, "reward": 0.8669579029083252, "reward_std": 0.24918241798877716, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03274257108569145, "sampling/sampling_logp_difference/max": 1.0835933685302734, "sampling/importance_sampling_ratio/min": 0.3383774161338806, "sampling/importance_sampling_ratio/mean": 1.0107687711715698, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3485318683087826, "clip_ratio/low_mean": 0.008140756515786052, "clip_ratio/low_min": 0.008140756515786052, "clip_ratio/high_mean": 0.012675640406087041, "clip_ratio/high_max": 0.012675640406087041, "clip_ratio/region_mean": 0.020816396921873093, "reward_total_mean": 0.8669579029083252, "reward_meter_mean": 0.9025046229362488, "reward_meter_std": 0.2555387020111084, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.8669579029083252, "reward_total_composite_std": 0.24918241798877716} {"timestamp_utc": "2026-04-12T03:09:48Z", "mode": "train", "global_step": 3004, "epoch": 0.12065710728200185, "loss": 0.0001, "grad_norm": 4.322078704833984, "learning_rate": 9.000000000000001e-07, "num_tokens": 6821058.0, "completions/mean_length": 128.75, "completions/min_length": 128.0, "completions/max_length": 130.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.75, "completions/min_terminated_length": 128.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.9970972537994385, "rewards/meter/std": 0.001955085899680853, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9970972537994385, "rewards/total_composite/std": 0.001955085899680853, "reward": 0.9970972537994385, "reward_std": 0.0019550775177776814, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021072760224342346, "sampling/sampling_logp_difference/max": 1.0552425384521484, "sampling/importance_sampling_ratio/min": 0.3481079638004303, "sampling/importance_sampling_ratio/mean": 1.0051413774490356, "sampling/importance_sampling_ratio/max": 1.4015967845916748, "entropy": 0.1789141520857811, "clip_ratio/low_mean": 0.002914663462433964, "clip_ratio/low_min": 0.002914663462433964, "clip_ratio/high_mean": 0.00777464872226119, "clip_ratio/high_max": 0.00777464872226119, "clip_ratio/region_mean": 0.010689312184695154, "reward_total_mean": 0.9970972537994385, "reward_meter_mean": 0.9970972537994385, "reward_meter_std": 0.001955085899680853, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9970972537994385, "reward_total_composite_std": 0.001955085899680853} {"timestamp_utc": "2026-04-12T03:09:55Z", "mode": "train", "global_step": 3005, "epoch": 0.1206972727637868, "loss": 0.0018, "grad_norm": 1.2387527227401733, "learning_rate": 8.96969696969697e-07, "num_tokens": 6824278.0, "completions/mean_length": 194.5, "completions/min_length": 190.0, "completions/max_length": 197.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 194.5, "completions/min_terminated_length": 190.0, "completions/max_terminated_length": 197.0, "rewards/meter/mean": 0.9990871548652649, "rewards/meter/std": 8.784286910668015e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990871548652649, "rewards/total_composite/std": 8.784286910668015e-05, "reward": 0.9990871548652649, "reward_std": 8.781959331827238e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03491505607962608, "sampling/sampling_logp_difference/max": 0.9021329879760742, "sampling/importance_sampling_ratio/min": 0.4057033956050873, "sampling/importance_sampling_ratio/mean": 1.0082441568374634, "sampling/importance_sampling_ratio/max": 1.4649758338928223, "entropy": 0.3335018455982208, "clip_ratio/low_mean": 0.009682861273176968, "clip_ratio/low_min": 0.009682861273176968, "clip_ratio/high_mean": 0.007038468029350042, "clip_ratio/high_max": 0.007038468029350042, "clip_ratio/region_mean": 0.01672132930252701, "reward_total_mean": 0.9990871548652649, "reward_meter_mean": 0.9990871548652649, "reward_meter_std": 8.784286910668015e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990871548652649, "reward_total_composite_std": 8.784286910668015e-05} {"timestamp_utc": "2026-04-12T03:09:59Z", "mode": "train", "global_step": 3006, "epoch": 0.12073743824557176, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 8.93939393939394e-07, "num_tokens": 6825838.0, "completions/mean_length": 32.0, "completions/min_length": 32.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "reward": 0.9992982149124146, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0001296020782319829, "sampling/sampling_logp_difference/max": 0.003267202526330948, "sampling/importance_sampling_ratio/min": 0.9980201125144958, "sampling/importance_sampling_ratio/mean": 1.0001120567321777, "sampling/importance_sampling_ratio/max": 1.003272533416748, "entropy": 0.0011012316681444645, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9992982149124146, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:10:05Z", "mode": "train", "global_step": 3007, "epoch": 0.12077760372735671, "loss": 0.002, "grad_norm": 3.9469776153564453, "learning_rate": 8.90909090909091e-07, "num_tokens": 6828661.0, "completions/mean_length": 167.875, "completions/min_length": 164.0, "completions/max_length": 173.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 167.875, "completions/min_terminated_length": 164.0, "completions/max_terminated_length": 173.0, "rewards/meter/mean": 0.9986657500267029, "rewards/meter/std": 0.0007131828460842371, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.9709138870239258, "rewards/total_composite/std": 0.05114205181598663, "reward": 0.9709138870239258, "reward_std": 0.05114204064011574, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04064127802848816, "sampling/sampling_logp_difference/max": 2.306015968322754, "sampling/importance_sampling_ratio/min": 0.09965749830007553, "sampling/importance_sampling_ratio/mean": 1.0033116340637207, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31089969351887703, "clip_ratio/low_mean": 0.006006060168147087, "clip_ratio/low_min": 0.006006060168147087, "clip_ratio/high_mean": 0.02804516162723303, "clip_ratio/high_max": 0.02804516162723303, "clip_ratio/region_mean": 0.034051221795380116, "reward_total_mean": 0.9709138870239258, "reward_meter_mean": 0.9986657500267029, "reward_meter_std": 0.0007131828460842371, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_total_composite_mean": 0.9709138870239258, "reward_total_composite_std": 0.05114205181598663} {"timestamp_utc": "2026-04-12T03:10:14Z", "mode": "train", "global_step": 3008, "epoch": 0.12081776920914167, "loss": 0.0014, "grad_norm": 1.8470708131790161, "learning_rate": 8.87878787878788e-07, "num_tokens": 6833680.0, "completions/mean_length": 380.375, "completions/min_length": 354.0, "completions/max_length": 391.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 380.375, "completions/min_terminated_length": 354.0, "completions/max_terminated_length": 391.0, "rewards/meter/mean": 0.9990596771240234, "rewards/meter/std": 0.00017261884931940585, "rewards/count_adherence/mean": 0.8863636255264282, "rewards/count_adherence/std": 0.04208274558186531, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9404239654541016, "rewards/repeat_penalty/std": 0.06556747853755951, "rewards/total_composite/mean": 0.8320455551147461, "rewards/total_composite/std": 0.06093808636069298, "reward": 0.8320455551147461, "reward_std": 0.060938093811273575, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.044413696974515915, "sampling/sampling_logp_difference/max": 1.220409870147705, "sampling/importance_sampling_ratio/min": 0.29510918259620667, "sampling/importance_sampling_ratio/mean": 1.011826992034912, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40278755873441696, "clip_ratio/low_mean": 0.008936886792071164, "clip_ratio/low_min": 0.008936886792071164, "clip_ratio/high_mean": 0.013912130380049348, "clip_ratio/high_max": 0.013912130380049348, "clip_ratio/region_mean": 0.02284901717212051, "reward_total_mean": 0.8320455551147461, "reward_meter_mean": 0.9990596771240234, "reward_meter_std": 0.00017261884931940585, "reward_count_adherence_mean": 0.8863636255264282, "reward_count_adherence_std": 0.04208274558186531, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9404239654541016, "reward_repeat_penalty_std": 0.06556747853755951, "reward_total_composite_mean": 0.8320455551147461, "reward_total_composite_std": 0.06093808636069298} {"timestamp_utc": "2026-04-12T03:10:19Z", "mode": "train", "global_step": 3009, "epoch": 0.12085793469092662, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 8.84848484848485e-07, "num_tokens": 6835408.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002200069575337693, "sampling/sampling_logp_difference/max": 0.0032224380411207676, "sampling/importance_sampling_ratio/min": 0.9967827200889587, "sampling/importance_sampling_ratio/mean": 1.000199556350708, "sampling/importance_sampling_ratio/max": 1.0028958320617676, "entropy": 0.0017017422651406378, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:10:23Z", "mode": "train", "global_step": 3010, "epoch": 0.12089810017271158, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 8.818181818181819e-07, "num_tokens": 6836936.0, "completions/mean_length": 32.0, "completions/min_length": 32.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "reward": 0.9992982149124146, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 4.44791694462765e-05, "sampling/sampling_logp_difference/max": 0.0007356764399446547, "sampling/importance_sampling_ratio/min": 0.9992645978927612, "sampling/importance_sampling_ratio/mean": 1.0000369548797607, "sampling/importance_sampling_ratio/max": 1.0006588697433472, "entropy": 0.0004426717623573495, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9992982149124146, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:10:29Z", "mode": "train", "global_step": 3011, "epoch": 0.12093826565449653, "loss": -0.0004, "grad_norm": 0.6899365782737732, "learning_rate": 8.787878787878788e-07, "num_tokens": 6839162.0, "completions/mean_length": 106.25, "completions/min_length": 105.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.25, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.999291181564331, "rewards/meter/std": 7.206718146335334e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999291181564331, "rewards/total_composite/std": 7.206718146335334e-05, "reward": 0.999291181564331, "reward_std": 7.208617898868397e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013669301755726337, "sampling/sampling_logp_difference/max": 0.952059268951416, "sampling/importance_sampling_ratio/min": 0.3859454393386841, "sampling/importance_sampling_ratio/mean": 1.0045570135116577, "sampling/importance_sampling_ratio/max": 1.7733577489852905, "entropy": 0.13793968316167593, "clip_ratio/low_mean": 0.002358490601181984, "clip_ratio/low_min": 0.002358490601181984, "clip_ratio/high_mean": 0.003494060132652521, "clip_ratio/high_max": 0.003494060132652521, "clip_ratio/region_mean": 0.005852550733834505, "reward_total_mean": 0.999291181564331, "reward_meter_mean": 0.999291181564331, "reward_meter_std": 7.206718146335334e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.999291181564331, "reward_total_composite_std": 7.206718146335334e-05} {"timestamp_utc": "2026-04-12T03:10:34Z", "mode": "train", "global_step": 3012, "epoch": 0.12097843113628148, "loss": 0.0008, "grad_norm": 3.873194694519043, "learning_rate": 8.757575757575758e-07, "num_tokens": 6841385.0, "completions/mean_length": 99.875, "completions/min_length": 99.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.875, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.9991791248321533, "rewards/meter/std": 0.00015561260806862265, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9741964340209961, "rewards/total_composite/std": 0.07060983777046204, "reward": 0.9741964340209961, "reward_std": 0.07060984522104263, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023857776075601578, "sampling/sampling_logp_difference/max": 2.5868468284606934, "sampling/importance_sampling_ratio/min": 0.07525696605443954, "sampling/importance_sampling_ratio/mean": 1.0067286491394043, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.159406297840178, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/high_mean": 0.008762876153923571, "clip_ratio/high_max": 0.008762876153923571, "clip_ratio/region_mean": 0.012550755054689944, "reward_total_mean": 0.9741964340209961, "reward_meter_mean": 0.9991791248321533, "reward_meter_std": 0.00015561260806862265, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9741964340209961, "reward_total_composite_std": 0.07060983777046204} {"timestamp_utc": "2026-04-12T03:10:39Z", "mode": "train", "global_step": 3013, "epoch": 0.12101859661806644, "loss": -0.0001, "grad_norm": 1.6813957691192627, "learning_rate": 8.727272727272728e-07, "num_tokens": 6843559.0, "completions/mean_length": 92.75, "completions/min_length": 92.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.75, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.9975481629371643, "rewards/meter/std": 0.0004655694356188178, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9975481629371643, "rewards/total_composite/std": 0.0004655694356188178, "reward": 0.9975481629371643, "reward_std": 0.00046558206668123603, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01227449532598257, "sampling/sampling_logp_difference/max": 1.0146255493164062, "sampling/importance_sampling_ratio/min": 0.3625381588935852, "sampling/importance_sampling_ratio/mean": 1.0000393390655518, "sampling/importance_sampling_ratio/max": 1.3771406412124634, "entropy": 0.06945714261382818, "clip_ratio/low_mean": 0.0013440860202535987, "clip_ratio/low_min": 0.0013440860202535987, "clip_ratio/high_mean": 0.00945243111345917, "clip_ratio/high_max": 0.00945243111345917, "clip_ratio/region_mean": 0.010796517133712769, "reward_total_mean": 0.9975481629371643, "reward_meter_mean": 0.9975481629371643, "reward_meter_std": 0.0004655694356188178, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9975481629371643, "reward_total_composite_std": 0.0004655694356188178} {"timestamp_utc": "2026-04-12T03:10:44Z", "mode": "train", "global_step": 3014, "epoch": 0.12105876209985139, "loss": -0.0004, "grad_norm": 1.1659772396087646, "learning_rate": 8.696969696969699e-07, "num_tokens": 6846028.0, "completions/mean_length": 141.625, "completions/min_length": 140.0, "completions/max_length": 143.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 141.625, "completions/min_terminated_length": 140.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.999117910861969, "rewards/meter/std": 0.00024422150454483926, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999117910861969, "rewards/total_composite/std": 0.00024422150454483926, "reward": 0.999117910861969, "reward_std": 0.0002442096301820129, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01364456582814455, "sampling/sampling_logp_difference/max": 0.7532033920288086, "sampling/importance_sampling_ratio/min": 0.47085580229759216, "sampling/importance_sampling_ratio/mean": 1.0043392181396484, "sampling/importance_sampling_ratio/max": 1.4682461023330688, "entropy": 0.17108981497585773, "clip_ratio/low_mean": 0.0017605633474886417, "clip_ratio/low_min": 0.0017605633474886417, "clip_ratio/high_mean": 0.007061248528771102, "clip_ratio/high_max": 0.007061248528771102, "clip_ratio/region_mean": 0.008821811876259744, "reward_total_mean": 0.999117910861969, "reward_meter_mean": 0.999117910861969, "reward_meter_std": 0.00024422150454483926, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.999117910861969, "reward_total_composite_std": 0.00024422150454483926} {"timestamp_utc": "2026-04-12T03:10:49Z", "mode": "train", "global_step": 3015, "epoch": 0.12109892758163635, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 8.666666666666668e-07, "num_tokens": 6847788.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 9.00184822967276e-05, "sampling/sampling_logp_difference/max": 0.0020202400628477335, "sampling/importance_sampling_ratio/min": 0.9989049434661865, "sampling/importance_sampling_ratio/mean": 1.0000646114349365, "sampling/importance_sampling_ratio/max": 1.0020222663879395, "entropy": 0.0011554057127796113, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:10:55Z", "mode": "train", "global_step": 3016, "epoch": 0.1211390930634213, "loss": 0.0068, "grad_norm": 3.0696825981140137, "learning_rate": 8.636363636363637e-07, "num_tokens": 6850815.0, "completions/mean_length": 166.375, "completions/min_length": 160.0, "completions/max_length": 170.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 166.375, "completions/min_terminated_length": 160.0, "completions/max_terminated_length": 170.0, "rewards/meter/mean": 0.9988635778427124, "rewards/meter/std": 0.0006956413271836936, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9988635778427124, "rewards/total_composite/std": 0.0006956413271836936, "reward": 0.9988635778427124, "reward_std": 0.0006956506404094398, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04618171229958534, "sampling/sampling_logp_difference/max": 2.1471054553985596, "sampling/importance_sampling_ratio/min": 0.11682181060314178, "sampling/importance_sampling_ratio/mean": 1.0038132667541504, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.32815663516521454, "clip_ratio/low_mean": 0.008875739760696888, "clip_ratio/low_min": 0.008875739760696888, "clip_ratio/high_mean": 0.02051854378078133, "clip_ratio/high_max": 0.02051854378078133, "clip_ratio/region_mean": 0.029394283541478217, "reward_total_mean": 0.9988635778427124, "reward_meter_mean": 0.9988635778427124, "reward_meter_std": 0.0006956413271836936, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9988635778427124, "reward_total_composite_std": 0.0006956413271836936} {"timestamp_utc": "2026-04-12T03:11:00Z", "mode": "train", "global_step": 3017, "epoch": 0.12117925854520625, "loss": 0.0083, "grad_norm": 2.4396045207977295, "learning_rate": 8.606060606060607e-07, "num_tokens": 6852962.0, "completions/mean_length": 97.375, "completions/min_length": 96.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.375, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.992943286895752, "rewards/meter/std": 0.0015025768661871552, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.992943286895752, "rewards/total_composite/std": 0.0015025768661871552, "reward": 0.992943286895752, "reward_std": 0.0015025660395622253, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026139099150896072, "sampling/sampling_logp_difference/max": 0.8849821090698242, "sampling/importance_sampling_ratio/min": 0.41272157430648804, "sampling/importance_sampling_ratio/mean": 1.0036437511444092, "sampling/importance_sampling_ratio/max": 1.6669363975524902, "entropy": 0.22009673342108727, "clip_ratio/low_mean": 0.0025641699321568012, "clip_ratio/low_min": 0.0025641699321568012, "clip_ratio/high_mean": 0.014189229113981128, "clip_ratio/high_max": 0.014189229113981128, "clip_ratio/region_mean": 0.01675339904613793, "reward_total_mean": 0.992943286895752, "reward_meter_mean": 0.992943286895752, "reward_meter_std": 0.0015025768661871552, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.992943286895752, "reward_total_composite_std": 0.0015025768661871552} {"timestamp_utc": "2026-04-12T03:11:10Z", "mode": "train", "global_step": 3018, "epoch": 0.12121942402699121, "loss": -0.0211, "grad_norm": 1.3273999691009521, "learning_rate": 8.575757575757576e-07, "num_tokens": 6857392.0, "completions/mean_length": 375.75, "completions/min_length": 348.0, "completions/max_length": 392.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 375.75, "completions/min_terminated_length": 348.0, "completions/max_terminated_length": 392.0, "rewards/meter/mean": 0.9991037845611572, "rewards/meter/std": 9.388383477926254e-05, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.04704993963241577, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9593868255615234, "rewards/repeat_penalty/std": 0.03774290531873703, "rewards/total_composite/mean": 0.8386253118515015, "rewards/total_composite/std": 0.054765164852142334, "reward": 0.8386253118515015, "reward_std": 0.05476517230272293, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04461507499217987, "sampling/sampling_logp_difference/max": 1.3247604370117188, "sampling/importance_sampling_ratio/min": 0.2658666670322418, "sampling/importance_sampling_ratio/mean": 1.012117624282837, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4122394025325775, "clip_ratio/low_mean": 0.009317906049545854, "clip_ratio/low_min": 0.009317906049545854, "clip_ratio/high_mean": 0.016257623909041286, "clip_ratio/high_max": 0.016257623909041286, "clip_ratio/region_mean": 0.02557552995858714, "reward_total_mean": 0.8386253118515015, "reward_meter_mean": 0.9991037845611572, "reward_meter_std": 9.388383477926254e-05, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.04704993963241577, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9593868255615234, "reward_repeat_penalty_std": 0.03774290531873703, "reward_total_composite_mean": 0.8386253118515015, "reward_total_composite_std": 0.054765164852142334} {"timestamp_utc": "2026-04-12T03:11:15Z", "mode": "train", "global_step": 3019, "epoch": 0.12125958950877616, "loss": -0.0003, "grad_norm": 0.24967576563358307, "learning_rate": 8.545454545454546e-07, "num_tokens": 6859104.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9973341226577759, "rewards/meter/std": 9.24646246858174e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973341226577759, "rewards/total_composite/std": 9.24646246858174e-06, "reward": 0.9973341226577759, "reward_std": 9.240244253305718e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.002222540322691202, "sampling/sampling_logp_difference/max": 0.16164398193359375, "sampling/importance_sampling_ratio/min": 0.8715769052505493, "sampling/importance_sampling_ratio/mean": 1.0005959272384644, "sampling/importance_sampling_ratio/max": 1.1754417419433594, "entropy": 0.021813111379742622, "clip_ratio/low_mean": 0.0020491802133619785, "clip_ratio/low_min": 0.0020491802133619785, "clip_ratio/high_mean": 0.0020491802133619785, "clip_ratio/high_max": 0.0020491802133619785, "clip_ratio/region_mean": 0.004098360426723957, "reward_total_mean": 0.9973341226577759, "reward_meter_mean": 0.9973341226577759, "reward_meter_std": 9.24646246858174e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9973341226577759, "reward_total_composite_std": 9.24646246858174e-06} {"timestamp_utc": "2026-04-12T03:11:20Z", "mode": "train", "global_step": 3020, "epoch": 0.12129975499056111, "loss": 0.0005, "grad_norm": 0.15372411906719208, "learning_rate": 8.515151515151515e-07, "num_tokens": 6861223.0, "completions/mean_length": 97.875, "completions/min_length": 97.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.875, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9994118809700012, "rewards/meter/std": 1.710540527710691e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994118809700012, "rewards/total_composite/std": 1.710540527710691e-05, "reward": 0.9994118809700012, "reward_std": 1.710999640636146e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006246503908187151, "sampling/sampling_logp_difference/max": 0.6895544528961182, "sampling/importance_sampling_ratio/min": 0.5627803802490234, "sampling/importance_sampling_ratio/mean": 1.00173020362854, "sampling/importance_sampling_ratio/max": 1.9928275346755981, "entropy": 0.037544872146099806, "clip_ratio/low_mean": 0.0012755101779475808, "clip_ratio/low_min": 0.0012755101779475808, "clip_ratio/high_mean": 0.0025641699321568012, "clip_ratio/high_max": 0.0025641699321568012, "clip_ratio/region_mean": 0.003839680110104382, "reward_total_mean": 0.9994118809700012, "reward_meter_mean": 0.9994118809700012, "reward_meter_std": 1.710540527710691e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994118809700012, "reward_total_composite_std": 1.710540527710691e-05} {"timestamp_utc": "2026-04-12T03:11:26Z", "mode": "train", "global_step": 3021, "epoch": 0.12133992047234607, "loss": 0.0012, "grad_norm": 1.7028465270996094, "learning_rate": 8.484848484848486e-07, "num_tokens": 6864084.0, "completions/mean_length": 177.625, "completions/min_length": 177.0, "completions/max_length": 179.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 177.625, "completions/min_terminated_length": 177.0, "completions/max_terminated_length": 179.0, "rewards/meter/mean": 0.9990875720977783, "rewards/meter/std": 0.00012965931091457605, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.9713343381881714, "rewards/total_composite/std": 0.05137185752391815, "reward": 0.9713343381881714, "reward_std": 0.05137185752391815, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018357736989855766, "sampling/sampling_logp_difference/max": 0.9017171859741211, "sampling/importance_sampling_ratio/min": 0.405872106552124, "sampling/importance_sampling_ratio/mean": 1.004658818244934, "sampling/importance_sampling_ratio/max": 1.6029589176177979, "entropy": 0.20351556316018105, "clip_ratio/low_mean": 0.0021146765793673694, "clip_ratio/low_min": 0.0021146765793673694, "clip_ratio/high_mean": 0.011962095857597888, "clip_ratio/high_max": 0.011962095857597888, "clip_ratio/region_mean": 0.014076772436965257, "reward_total_mean": 0.9713343381881714, "reward_meter_mean": 0.9990875720977783, "reward_meter_std": 0.00012965931091457605, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_total_composite_mean": 0.9713343381881714, "reward_total_composite_std": 0.05137185752391815} {"timestamp_utc": "2026-04-12T03:11:31Z", "mode": "train", "global_step": 3022, "epoch": 0.12138008595413102, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 8.454545454545456e-07, "num_tokens": 6865556.0, "completions/mean_length": 32.0, "completions/min_length": 32.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "reward": 0.9992982149124146, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 5.831335874972865e-05, "sampling/sampling_logp_difference/max": 0.0005344097153283656, "sampling/importance_sampling_ratio/min": 0.9995866417884827, "sampling/importance_sampling_ratio/mean": 1.0000536441802979, "sampling/importance_sampling_ratio/max": 1.0005345344543457, "entropy": 0.00045082847282174043, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9992982149124146, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:11:36Z", "mode": "train", "global_step": 3023, "epoch": 0.12142025143591598, "loss": 0.0024, "grad_norm": 3.5771987438201904, "learning_rate": 8.424242424242425e-07, "num_tokens": 6867729.0, "completions/mean_length": 102.625, "completions/min_length": 101.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.625, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.9883763194084167, "rewards/meter/std": 0.009592375718057156, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9883763194084167, "rewards/total_composite/std": 0.009592375718057156, "reward": 0.9883763194084167, "reward_std": 0.009592383168637753, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.039184827357530594, "sampling/sampling_logp_difference/max": 0.941868782043457, "sampling/importance_sampling_ratio/min": 0.38989853858947754, "sampling/importance_sampling_ratio/mean": 1.0047192573547363, "sampling/importance_sampling_ratio/max": 1.92518150806427, "entropy": 0.359358724206686, "clip_ratio/low_mean": 0.012183514423668385, "clip_ratio/low_min": 0.012183514423668385, "clip_ratio/high_mean": 0.017015473917126656, "clip_ratio/high_max": 0.017015473917126656, "clip_ratio/region_mean": 0.02919898834079504, "reward_total_mean": 0.9883763194084167, "reward_meter_mean": 0.9883763194084167, "reward_meter_std": 0.009592375718057156, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9883763194084167, "reward_total_composite_std": 0.009592375718057156} {"timestamp_utc": "2026-04-12T03:11:40Z", "mode": "train", "global_step": 3024, "epoch": 0.12146041691770093, "loss": -0.0002, "grad_norm": 0.03466307371854782, "learning_rate": 8.393939393939395e-07, "num_tokens": 6869889.0, "completions/mean_length": 98.0, "completions/min_length": 98.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.0, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9994069337844849, "rewards/meter/std": 1.776606950443238e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994069337844849, "rewards/total_composite/std": 1.776606950443238e-06, "reward": 0.9994069337844849, "reward_std": 1.775535338310874e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005284504033625126, "sampling/sampling_logp_difference/max": 0.844066858291626, "sampling/importance_sampling_ratio/min": 0.4993957281112671, "sampling/importance_sampling_ratio/mean": 1.0014625787734985, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.031393167562782764, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0012755101779475808, "clip_ratio/high_max": 0.0012755101779475808, "clip_ratio/region_mean": 0.0012755101779475808, "reward_total_mean": 0.9994069337844849, "reward_meter_mean": 0.9994069337844849, "reward_meter_std": 1.776606950443238e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994069337844849, "reward_total_composite_std": 1.776606950443238e-06} {"timestamp_utc": "2026-04-12T03:11:46Z", "mode": "train", "global_step": 3025, "epoch": 0.12150058239948588, "loss": -0.0004, "grad_norm": 0.8292710185050964, "learning_rate": 8.363636363636364e-07, "num_tokens": 6872288.0, "completions/mean_length": 131.875, "completions/min_length": 131.0, "completions/max_length": 132.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.875, "completions/min_terminated_length": 131.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.99941086769104, "rewards/meter/std": 6.445148028433323e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.99941086769104, "rewards/total_composite/std": 6.445148028433323e-05, "reward": 0.99941086769104, "reward_std": 6.445525650633499e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008050598204135895, "sampling/sampling_logp_difference/max": 0.6427021026611328, "sampling/importance_sampling_ratio/min": 0.5258695483207703, "sampling/importance_sampling_ratio/mean": 1.0006563663482666, "sampling/importance_sampling_ratio/max": 1.33772611618042, "entropy": 0.0625457102432847, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.006628788076341152, "clip_ratio/high_max": 0.006628788076341152, "clip_ratio/region_mean": 0.006628788076341152, "reward_total_mean": 0.99941086769104, "reward_meter_mean": 0.99941086769104, "reward_meter_std": 6.445148028433323e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.99941086769104, "reward_total_composite_std": 6.445148028433323e-05} {"timestamp_utc": "2026-04-12T03:11:51Z", "mode": "train", "global_step": 3026, "epoch": 0.12154074788127084, "loss": 0.0631, "grad_norm": 6.055728435516357, "learning_rate": 8.333333333333333e-07, "num_tokens": 6874658.0, "completions/mean_length": 116.25, "completions/min_length": 107.0, "completions/max_length": 127.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.25, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.9515209197998047, "rewards/meter/std": 0.023815272375941277, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9633838534355164, "rewards/repeat_penalty/std": 0.05091821402311325, "rewards/total_composite/mean": 0.8265341520309448, "rewards/total_composite/std": 0.12215512990951538, "reward": 0.8265341520309448, "reward_std": 0.12215511500835419, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06652670353651047, "sampling/sampling_logp_difference/max": 2.105895757675171, "sampling/importance_sampling_ratio/min": 0.1217365711927414, "sampling/importance_sampling_ratio/mean": 0.9996247291564941, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.30979629419744015, "clip_ratio/low_mean": 0.023373008705675602, "clip_ratio/low_min": 0.023373008705675602, "clip_ratio/high_mean": 0.01498432899825275, "clip_ratio/high_max": 0.01498432899825275, "clip_ratio/region_mean": 0.03835733770392835, "reward_total_mean": 0.8265341520309448, "reward_meter_mean": 0.9515209197998047, "reward_meter_std": 0.023815272375941277, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9633838534355164, "reward_repeat_penalty_std": 0.05091821402311325, "reward_total_composite_mean": 0.8265341520309448, "reward_total_composite_std": 0.12215512990951538} {"timestamp_utc": "2026-04-12T03:11:56Z", "mode": "train", "global_step": 3027, "epoch": 0.12158091336305579, "loss": -0.0003, "grad_norm": 1.3954490423202515, "learning_rate": 8.303030303030303e-07, "num_tokens": 6876567.0, "completions/mean_length": 78.625, "completions/min_length": 78.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.625, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9990073442459106, "rewards/meter/std": 0.000114113834570162, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990073442459106, "rewards/total_composite/std": 0.000114113834570162, "reward": 0.9990073442459106, "reward_std": 0.00011411593004595488, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02237360179424286, "sampling/sampling_logp_difference/max": 0.7234420776367188, "sampling/importance_sampling_ratio/min": 0.48507970571517944, "sampling/importance_sampling_ratio/mean": 1.00530207157135, "sampling/importance_sampling_ratio/max": 1.6200668811798096, "entropy": 0.18712522648274899, "clip_ratio/low_mean": 0.007972249411977828, "clip_ratio/low_min": 0.007972249411977828, "clip_ratio/high_mean": 0.014263797434978187, "clip_ratio/high_max": 0.014263797434978187, "clip_ratio/region_mean": 0.022236046846956015, "reward_total_mean": 0.9990073442459106, "reward_meter_mean": 0.9990073442459106, "reward_meter_std": 0.000114113834570162, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990073442459106, "reward_total_composite_std": 0.000114113834570162} {"timestamp_utc": "2026-04-12T03:12:00Z", "mode": "train", "global_step": 3028, "epoch": 0.12162107884484075, "loss": -0.0, "grad_norm": 0.22835104167461395, "learning_rate": 8.272727272727274e-07, "num_tokens": 6878303.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9973360300064087, "rewards/meter/std": 8.113272997434251e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973360300064087, "rewards/total_composite/std": 8.113272997434251e-06, "reward": 0.9973360300064087, "reward_std": 8.10424353403505e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0027969391085207462, "sampling/sampling_logp_difference/max": 0.7328996658325195, "sampling/importance_sampling_ratio/min": 0.48051363229751587, "sampling/importance_sampling_ratio/mean": 0.9999331831932068, "sampling/importance_sampling_ratio/max": 1.1025102138519287, "entropy": 0.015828296658582985, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9973360300064087, "reward_meter_mean": 0.9973360300064087, "reward_meter_std": 8.113272997434251e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9973360300064087, "reward_total_composite_std": 8.113272997434251e-06} {"timestamp_utc": "2026-04-12T03:12:05Z", "mode": "train", "global_step": 3029, "epoch": 0.1216612443266257, "loss": 0.0018, "grad_norm": 0.5074459314346313, "learning_rate": 8.242424242424244e-07, "num_tokens": 6880686.0, "completions/mean_length": 107.875, "completions/min_length": 107.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.875, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.9992820620536804, "rewards/meter/std": 4.010711199953221e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992820620536804, "rewards/total_composite/std": 4.010711199953221e-05, "reward": 0.9992820620536804, "reward_std": 4.012084173155017e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018780170008540154, "sampling/sampling_logp_difference/max": 1.2216401100158691, "sampling/importance_sampling_ratio/min": 0.29474636912345886, "sampling/importance_sampling_ratio/mean": 1.0014318227767944, "sampling/importance_sampling_ratio/max": 1.7053853273391724, "entropy": 0.1494951182976365, "clip_ratio/low_mean": 0.008082556771114469, "clip_ratio/low_min": 0.008082556771114469, "clip_ratio/high_mean": 0.008112668758258224, "clip_ratio/high_max": 0.008112668758258224, "clip_ratio/region_mean": 0.016195225529372692, "reward_total_mean": 0.9992820620536804, "reward_meter_mean": 0.9992820620536804, "reward_meter_std": 4.010711199953221e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992820620536804, "reward_total_composite_std": 4.010711199953221e-05} {"timestamp_utc": "2026-04-12T03:12:13Z", "mode": "train", "global_step": 3030, "epoch": 0.12170140980841065, "loss": -0.0054, "grad_norm": 1.8386452198028564, "learning_rate": 8.212121212121213e-07, "num_tokens": 6885074.0, "completions/mean_length": 334.5, "completions/min_length": 323.0, "completions/max_length": 343.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 334.5, "completions/min_terminated_length": 323.0, "completions/max_terminated_length": 343.0, "rewards/meter/mean": 0.9939321279525757, "rewards/meter/std": 0.0059409006498754025, "rewards/count_adherence/mean": 0.9090909361839294, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9621710777282715, "rewards/repeat_penalty/std": 0.08768600970506668, "rewards/total_composite/mean": 0.8695409297943115, "rewards/total_composite/std": 0.08088699728250504, "reward": 0.8695409297943115, "reward_std": 0.08088699728250504, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.054148994386196136, "sampling/sampling_logp_difference/max": 2.918501138687134, "sampling/importance_sampling_ratio/min": 0.05401458963751793, "sampling/importance_sampling_ratio/mean": 1.0112972259521484, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4611641392111778, "clip_ratio/low_mean": 0.0044835947919636965, "clip_ratio/low_min": 0.0044835947919636965, "clip_ratio/high_mean": 0.033679026179015636, "clip_ratio/high_max": 0.033679026179015636, "clip_ratio/region_mean": 0.03816262097097933, "reward_total_mean": 0.8695409297943115, "reward_meter_mean": 0.9939321279525757, "reward_meter_std": 0.0059409006498754025, "reward_count_adherence_mean": 0.9090909361839294, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9621710777282715, "reward_repeat_penalty_std": 0.08768600970506668, "reward_total_composite_mean": 0.8695409297943115, "reward_total_composite_std": 0.08088699728250504} {"timestamp_utc": "2026-04-12T03:12:19Z", "mode": "train", "global_step": 3031, "epoch": 0.12174157529019561, "loss": -0.0067, "grad_norm": 2.3149688243865967, "learning_rate": 8.181818181818182e-07, "num_tokens": 6887904.0, "completions/mean_length": 154.75, "completions/min_length": 150.0, "completions/max_length": 158.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 154.75, "completions/min_terminated_length": 150.0, "completions/max_terminated_length": 158.0, "rewards/meter/mean": 0.9982709884643555, "rewards/meter/std": 0.0024356048088520765, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9982709884643555, "rewards/total_composite/std": 0.0024356048088520765, "reward": 0.9982709884643555, "reward_std": 0.0024356169160455465, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03701268509030342, "sampling/sampling_logp_difference/max": 1.422220230102539, "sampling/importance_sampling_ratio/min": 0.24117796123027802, "sampling/importance_sampling_ratio/mean": 1.009082555770874, "sampling/importance_sampling_ratio/max": 1.942898154258728, "entropy": 0.33194397762417793, "clip_ratio/low_mean": 0.0033333334140479565, "clip_ratio/low_min": 0.0033333334140479565, "clip_ratio/high_mean": 0.02090477745514363, "clip_ratio/high_max": 0.02090477745514363, "clip_ratio/region_mean": 0.024238110869191587, "reward_total_mean": 0.9982709884643555, "reward_meter_mean": 0.9982709884643555, "reward_meter_std": 0.0024356048088520765, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9982709884643555, "reward_total_composite_std": 0.0024356048088520765} {"timestamp_utc": "2026-04-12T03:12:24Z", "mode": "train", "global_step": 3032, "epoch": 0.12178174077198056, "loss": -0.0001, "grad_norm": 0.31778454780578613, "learning_rate": 8.151515151515152e-07, "num_tokens": 6889852.0, "completions/mean_length": 66.5, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.5, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981399774551392, "rewards/meter/std": 1.4936525076336693e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981399774551392, "rewards/total_composite/std": 1.4936525076336693e-05, "reward": 0.9981399774551392, "reward_std": 1.4939930224500131e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007410375867038965, "sampling/sampling_logp_difference/max": 0.7496395111083984, "sampling/importance_sampling_ratio/min": 0.4725368618965149, "sampling/importance_sampling_ratio/mean": 0.9997544884681702, "sampling/importance_sampling_ratio/max": 1.3944506645202637, "entropy": 0.05587592348456383, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/high_mean": 0.007490954245440662, "clip_ratio/high_max": 0.007490954245440662, "clip_ratio/region_mean": 0.009384893695823848, "reward_total_mean": 0.9981399774551392, "reward_meter_mean": 0.9981399774551392, "reward_meter_std": 1.4936525076336693e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981399774551392, "reward_total_composite_std": 1.4936525076336693e-05} {"timestamp_utc": "2026-04-12T03:12:30Z", "mode": "train", "global_step": 3033, "epoch": 0.12182190625376552, "loss": 0.0028, "grad_norm": 2.056260347366333, "learning_rate": 8.121212121212121e-07, "num_tokens": 6892787.0, "completions/mean_length": 186.875, "completions/min_length": 183.0, "completions/max_length": 189.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 186.875, "completions/min_terminated_length": 183.0, "completions/max_terminated_length": 189.0, "rewards/meter/mean": 0.9618879556655884, "rewards/meter/std": 0.06856533139944077, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9090909361839294, "rewards/repeat_penalty/std": 0.0971859022974968, "rewards/total_composite/mean": 0.8737800121307373, "rewards/total_composite/std": 0.10824880003929138, "reward": 0.8737800121307373, "reward_std": 0.10824880748987198, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.034711167216300964, "sampling/sampling_logp_difference/max": 1.0839767456054688, "sampling/importance_sampling_ratio/min": 0.33824771642684937, "sampling/importance_sampling_ratio/mean": 1.0074325799942017, "sampling/importance_sampling_ratio/max": 1.8432459831237793, "entropy": 0.36450883373618126, "clip_ratio/low_mean": 0.006038419320248067, "clip_ratio/low_min": 0.006038419320248067, "clip_ratio/high_mean": 0.016598281217738986, "clip_ratio/high_max": 0.016598281217738986, "clip_ratio/region_mean": 0.022636700537987053, "reward_total_mean": 0.8737800121307373, "reward_meter_mean": 0.9618879556655884, "reward_meter_std": 0.06856533139944077, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9090909361839294, "reward_repeat_penalty_std": 0.0971859022974968, "reward_total_composite_mean": 0.8737800121307373, "reward_total_composite_std": 0.10824880003929138} {"timestamp_utc": "2026-04-12T03:12:34Z", "mode": "train", "global_step": 3034, "epoch": 0.12186207173555047, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 8.09090909090909e-07, "num_tokens": 6894235.0, "completions/mean_length": 28.0, "completions/min_length": 28.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.0, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9943599104881287, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9943599104881287, "rewards/total_composite/std": 0.0, "reward": 0.9943599104881287, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0001638415560591966, "sampling/sampling_logp_difference/max": 0.004102423787117004, "sampling/importance_sampling_ratio/min": 0.9988238215446472, "sampling/importance_sampling_ratio/mean": 1.000151515007019, "sampling/importance_sampling_ratio/max": 1.0041109323501587, "entropy": 0.0012948041403433308, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9943599104881287, "reward_meter_mean": 0.9943599104881287, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9943599104881287, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:12:38Z", "mode": "train", "global_step": 3035, "epoch": 0.12190223721733542, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 8.060606060606062e-07, "num_tokens": 6895891.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002374651812715456, "sampling/sampling_logp_difference/max": 0.0032967175357043743, "sampling/importance_sampling_ratio/min": 0.9998509287834167, "sampling/importance_sampling_ratio/mean": 1.0002354383468628, "sampling/importance_sampling_ratio/max": 1.0033022165298462, "entropy": 0.0020439386717043817, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:12:42Z", "mode": "train", "global_step": 3036, "epoch": 0.12194240269912038, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 8.030303030303031e-07, "num_tokens": 6897435.0, "completions/mean_length": 29.0, "completions/min_length": 29.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9957436919212341, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957436919212341, "rewards/total_composite/std": 0.0, "reward": 0.9957436919212341, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00020185003813821822, "sampling/sampling_logp_difference/max": 0.008230682462453842, "sampling/importance_sampling_ratio/min": 0.9977694749832153, "sampling/importance_sampling_ratio/mean": 1.0001778602600098, "sampling/importance_sampling_ratio/max": 1.0082646608352661, "entropy": 0.0024824678112054244, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9957436919212341, "reward_meter_mean": 0.9957436919212341, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9957436919212341, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:12:47Z", "mode": "train", "global_step": 3037, "epoch": 0.12198256818090533, "loss": 0.0, "grad_norm": 4.432851791381836, "learning_rate": 8.000000000000001e-07, "num_tokens": 6899483.0, "completions/mean_length": 104.0, "completions/min_length": 102.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 104.0, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.9864717721939087, "rewards/meter/std": 0.024548275396227837, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9615679979324341, "rewards/total_composite/std": 0.07080353796482086, "reward": 0.9615679979324341, "reward_std": 0.07080353796482086, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03678695112466812, "sampling/sampling_logp_difference/max": 1.2483813762664795, "sampling/importance_sampling_ratio/min": 0.2869689166545868, "sampling/importance_sampling_ratio/mean": 1.0080146789550781, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2963132858276367, "clip_ratio/low_mean": 0.0012254902394488454, "clip_ratio/low_min": 0.0012254902394488454, "clip_ratio/high_mean": 0.023894709534943104, "clip_ratio/high_max": 0.023894709534943104, "clip_ratio/region_mean": 0.02512019977439195, "reward_total_mean": 0.9615679979324341, "reward_meter_mean": 0.9864717721939087, "reward_meter_std": 0.024548275396227837, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9615679979324341, "reward_total_composite_std": 0.07080353796482086} {"timestamp_utc": "2026-04-12T03:12:57Z", "mode": "train", "global_step": 3038, "epoch": 0.12202273366269029, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.96969696969697e-07, "num_tokens": 6900995.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.933377742767334, "rewards/meter/std": 0.09035933762788773, "rewards/count_adherence/mean": 0.6749999523162842, "rewards/count_adherence/std": 0.02672613225877285, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9716880321502686, "rewards/repeat_penalty/std": 0.01748696342110634, "rewards/total_composite/mean": 0.6106069087982178, "rewards/total_composite/std": 0.045595090836286545, "reward": 0.6106069087982178, "reward_std": 0.045595090836286545, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.6106069087982178, "reward_meter_mean": 0.933377742767334, "reward_meter_std": 0.09035933762788773, "reward_count_adherence_mean": 0.6749999523162842, "reward_count_adherence_std": 0.02672613225877285, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9716880321502686, "reward_repeat_penalty_std": 0.01748696342110634, "reward_total_composite_mean": 0.6106069087982178, "reward_total_composite_std": 0.045595090836286545} {"timestamp_utc": "2026-04-12T03:13:04Z", "mode": "train", "global_step": 3039, "epoch": 0.12206289914447524, "loss": 0.0015, "grad_norm": 1.2539805173873901, "learning_rate": 7.939393939393939e-07, "num_tokens": 6904381.0, "completions/mean_length": 233.25, "completions/min_length": 230.0, "completions/max_length": 238.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 233.25, "completions/min_terminated_length": 230.0, "completions/max_terminated_length": 238.0, "rewards/meter/mean": 0.9988183975219727, "rewards/meter/std": 0.000870219839271158, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9886363744735718, "rewards/repeat_penalty/std": 0.03214120864868164, "rewards/total_composite/mean": 0.987464189529419, "rewards/total_composite/std": 0.03198371082544327, "reward": 0.987464189529419, "reward_std": 0.031983714550733566, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03573669493198395, "sampling/sampling_logp_difference/max": 1.3547391891479492, "sampling/importance_sampling_ratio/min": 0.25801458954811096, "sampling/importance_sampling_ratio/mean": 1.012012243270874, "sampling/importance_sampling_ratio/max": 1.7767890691757202, "entropy": 0.3743705749511719, "clip_ratio/low_mean": 0.0021551724057644606, "clip_ratio/low_min": 0.0021551724057644606, "clip_ratio/high_mean": 0.022452297853305936, "clip_ratio/high_max": 0.022452297853305936, "clip_ratio/region_mean": 0.024607470259070396, "reward_total_mean": 0.987464189529419, "reward_meter_mean": 0.9988183975219727, "reward_meter_std": 0.000870219839271158, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9886363744735718, "reward_repeat_penalty_std": 0.03214120864868164, "reward_total_composite_mean": 0.987464189529419, "reward_total_composite_std": 0.03198371082544327} {"timestamp_utc": "2026-04-12T03:13:09Z", "mode": "train", "global_step": 3040, "epoch": 0.1221030646262602, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.909090909090909e-07, "num_tokens": 6906301.0, "completions/mean_length": 80.0, "completions/min_length": 80.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.0, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "reward": 0.7190229296684265, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002939449332188815, "sampling/sampling_logp_difference/max": 0.007810796611011028, "sampling/importance_sampling_ratio/min": 0.9997765421867371, "sampling/importance_sampling_ratio/mean": 1.000292420387268, "sampling/importance_sampling_ratio/max": 1.0078413486480713, "entropy": 0.00237001696950756, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7190229296684265, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:13:13Z", "mode": "train", "global_step": 3041, "epoch": 0.12214323010804515, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.878787878787879e-07, "num_tokens": 6907797.0, "completions/mean_length": 29.0, "completions/min_length": 29.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9957436919212341, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957436919212341, "rewards/total_composite/std": 0.0, "reward": 0.9957436919212341, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00019611754396464676, "sampling/sampling_logp_difference/max": 0.007875919342041016, "sampling/importance_sampling_ratio/min": 0.9921550154685974, "sampling/importance_sampling_ratio/mean": 1.0001239776611328, "sampling/importance_sampling_ratio/max": 1.0073140859603882, "entropy": 0.0016474624135298654, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9957436919212341, "reward_meter_mean": 0.9957436919212341, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9957436919212341, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:13:17Z", "mode": "train", "global_step": 3042, "epoch": 0.1221833955898301, "loss": -0.0003, "grad_norm": 0.7192491292953491, "learning_rate": 7.84848484848485e-07, "num_tokens": 6909726.0, "completions/mean_length": 71.125, "completions/min_length": 71.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.125, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9994059801101685, "rewards/meter/std": 3.530035974108614e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994059801101685, "rewards/total_composite/std": 3.530035974108614e-05, "reward": 0.9994059801101685, "reward_std": 3.531122638378292e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008167088031768799, "sampling/sampling_logp_difference/max": 0.9387979507446289, "sampling/importance_sampling_ratio/min": 0.39109766483306885, "sampling/importance_sampling_ratio/mean": 1.0006753206253052, "sampling/importance_sampling_ratio/max": 1.1902652978897095, "entropy": 0.0695438552647829, "clip_ratio/low_mean": 0.0034966744715347886, "clip_ratio/low_min": 0.0034966744715347886, "clip_ratio/high_mean": 0.0017605633474886417, "clip_ratio/high_max": 0.0017605633474886417, "clip_ratio/region_mean": 0.00525723781902343, "reward_total_mean": 0.9994059801101685, "reward_meter_mean": 0.9994059801101685, "reward_meter_std": 3.530035974108614e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994059801101685, "reward_total_composite_std": 3.530035974108614e-05} {"timestamp_utc": "2026-04-12T03:13:22Z", "mode": "train", "global_step": 3043, "epoch": 0.12222356107161506, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.818181818181819e-07, "num_tokens": 6911286.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002015612117247656, "sampling/sampling_logp_difference/max": 0.002510334365069866, "sampling/importance_sampling_ratio/min": 0.9998286962509155, "sampling/importance_sampling_ratio/mean": 1.0002003908157349, "sampling/importance_sampling_ratio/max": 1.0025135278701782, "entropy": 0.0016416620055679232, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:13:26Z", "mode": "train", "global_step": 3044, "epoch": 0.12226372655340001, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.787878787878788e-07, "num_tokens": 6913358.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00012283638352528214, "sampling/sampling_logp_difference/max": 0.004916388541460037, "sampling/importance_sampling_ratio/min": 0.9950957298278809, "sampling/importance_sampling_ratio/mean": 1.0000611543655396, "sampling/importance_sampling_ratio/max": 1.0022180080413818, "entropy": 0.0007720192006672733, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:13:31Z", "mode": "train", "global_step": 3045, "epoch": 0.12230389203518496, "loss": -0.0027, "grad_norm": 1.1839877367019653, "learning_rate": 7.757575757575758e-07, "num_tokens": 6915686.0, "completions/mean_length": 107.0, "completions/min_length": 105.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.0, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.9992265701293945, "rewards/meter/std": 0.0002075855591101572, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992265701293945, "rewards/total_composite/std": 0.0002075855591101572, "reward": 0.9992265701293945, "reward_std": 0.00020759036124218255, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01808370277285576, "sampling/sampling_logp_difference/max": 1.08237886428833, "sampling/importance_sampling_ratio/min": 0.33878862857818604, "sampling/importance_sampling_ratio/mean": 1.005001187324524, "sampling/importance_sampling_ratio/max": 1.8381428718566895, "entropy": 0.13631522469222546, "clip_ratio/low_mean": 0.005886243423447013, "clip_ratio/low_min": 0.005886243423447013, "clip_ratio/high_mean": 0.01052544778212905, "clip_ratio/high_max": 0.01052544778212905, "clip_ratio/region_mean": 0.016411691205576062, "reward_total_mean": 0.9992265701293945, "reward_meter_mean": 0.9992265701293945, "reward_meter_std": 0.0002075855591101572, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992265701293945, "reward_total_composite_std": 0.0002075855591101572} {"timestamp_utc": "2026-04-12T03:13:36Z", "mode": "train", "global_step": 3046, "epoch": 0.12234405751696992, "loss": -0.0018, "grad_norm": 1.057833194732666, "learning_rate": 7.727272727272727e-07, "num_tokens": 6917493.0, "completions/mean_length": 67.875, "completions/min_length": 67.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.875, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9994407892227173, "rewards/meter/std": 6.79898148518987e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994407892227173, "rewards/total_composite/std": 6.79898148518987e-05, "reward": 0.9994407892227173, "reward_std": 6.798980029998347e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0057174162939190865, "sampling/sampling_logp_difference/max": 0.2612924575805664, "sampling/importance_sampling_ratio/min": 0.7700556516647339, "sampling/importance_sampling_ratio/mean": 1.0019086599349976, "sampling/importance_sampling_ratio/max": 1.2558224201202393, "entropy": 0.04532818449661136, "clip_ratio/low_mean": 0.0018382353009656072, "clip_ratio/low_min": 0.0018382353009656072, "clip_ratio/high_mean": 0.0018382353009656072, "clip_ratio/high_max": 0.0018382353009656072, "clip_ratio/region_mean": 0.0036764706019312143, "reward_total_mean": 0.9994407892227173, "reward_meter_mean": 0.9994407892227173, "reward_meter_std": 6.79898148518987e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994407892227173, "reward_total_composite_std": 6.79898148518987e-05} {"timestamp_utc": "2026-04-12T03:13:42Z", "mode": "train", "global_step": 3047, "epoch": 0.12238422299875487, "loss": 0.005, "grad_norm": 1.7859693765640259, "learning_rate": 7.696969696969698e-07, "num_tokens": 6920501.0, "completions/mean_length": 200.0, "completions/min_length": 198.0, "completions/max_length": 202.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 200.0, "completions/min_terminated_length": 198.0, "completions/max_terminated_length": 202.0, "rewards/meter/mean": 0.9991546869277954, "rewards/meter/std": 0.0004873749567195773, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9090908765792847, "rewards/repeat_penalty/std": 0.06872081756591797, "rewards/total_composite/mean": 0.9083393812179565, "rewards/total_composite/std": 0.06891340762376785, "reward": 0.9083393812179565, "reward_std": 0.06891340017318726, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018795771524310112, "sampling/sampling_logp_difference/max": 1.3369550704956055, "sampling/importance_sampling_ratio/min": 0.2626442015171051, "sampling/importance_sampling_ratio/mean": 1.0054200887680054, "sampling/importance_sampling_ratio/max": 1.687934398651123, "entropy": 0.17142035998404026, "clip_ratio/low_mean": 0.005625279794912785, "clip_ratio/low_min": 0.005625279794912785, "clip_ratio/high_mean": 0.005637656955514103, "clip_ratio/high_max": 0.005637656955514103, "clip_ratio/region_mean": 0.011262936750426888, "reward_total_mean": 0.9083393812179565, "reward_meter_mean": 0.9991546869277954, "reward_meter_std": 0.0004873749567195773, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9090908765792847, "reward_repeat_penalty_std": 0.06872081756591797, "reward_total_composite_mean": 0.9083393812179565, "reward_total_composite_std": 0.06891340762376785} {"timestamp_utc": "2026-04-12T03:13:46Z", "mode": "train", "global_step": 3048, "epoch": 0.12242438848053983, "loss": -0.0015, "grad_norm": 5.802453994750977, "learning_rate": 7.666666666666667e-07, "num_tokens": 6921887.0, "completions/mean_length": 34.25, "completions/min_length": 34.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.25, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9926164150238037, "rewards/meter/std": 0.0012579105095937848, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9926164150238037, "rewards/total_composite/std": 0.0012579105095937848, "reward": 0.9926164150238037, "reward_std": 0.001257919822819531, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014383560046553612, "sampling/sampling_logp_difference/max": 1.1232624053955078, "sampling/importance_sampling_ratio/min": 0.325217068195343, "sampling/importance_sampling_ratio/mean": 0.9971778988838196, "sampling/importance_sampling_ratio/max": 1.4339221715927124, "entropy": 0.09313629940152168, "clip_ratio/low_mean": 0.0035714285913854837, "clip_ratio/low_min": 0.0035714285913854837, "clip_ratio/high_mean": 0.0035714285913854837, "clip_ratio/high_max": 0.0035714285913854837, "clip_ratio/region_mean": 0.0071428571827709675, "reward_total_mean": 0.9926164150238037, "reward_meter_mean": 0.9926164150238037, "reward_meter_std": 0.0012579105095937848, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9926164150238037, "reward_total_composite_std": 0.0012579105095937848} {"timestamp_utc": "2026-04-12T03:13:51Z", "mode": "train", "global_step": 3049, "epoch": 0.12246455396232478, "loss": -0.0064, "grad_norm": 3.5305817127227783, "learning_rate": 7.636363636363637e-07, "num_tokens": 6924412.0, "completions/mean_length": 137.625, "completions/min_length": 134.0, "completions/max_length": 141.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 137.625, "completions/min_terminated_length": 134.0, "completions/max_terminated_length": 141.0, "rewards/meter/mean": 0.9496752023696899, "rewards/meter/std": 0.0848766416311264, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9285714626312256, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.8786302804946899, "rewards/total_composite/std": 0.0753149688243866, "reward": 0.8786302804946899, "reward_std": 0.0753149539232254, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03166425973176956, "sampling/sampling_logp_difference/max": 1.181009292602539, "sampling/importance_sampling_ratio/min": 0.3069687485694885, "sampling/importance_sampling_ratio/mean": 1.0010079145431519, "sampling/importance_sampling_ratio/max": 1.5370413064956665, "entropy": 0.2945951074361801, "clip_ratio/low_mean": 0.011909950990229845, "clip_ratio/low_min": 0.011909950990229845, "clip_ratio/high_mean": 0.01253445539623499, "clip_ratio/high_max": 0.01253445539623499, "clip_ratio/region_mean": 0.024444406386464834, "reward_total_mean": 0.8786302804946899, "reward_meter_mean": 0.9496752023696899, "reward_meter_std": 0.0848766416311264, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9285714626312256, "reward_repeat_penalty_std": 0.07636035233736038, "reward_total_composite_mean": 0.8786302804946899, "reward_total_composite_std": 0.0753149688243866} {"timestamp_utc": "2026-04-12T03:13:58Z", "mode": "train", "global_step": 3050, "epoch": 0.12250471944410973, "loss": -0.0011, "grad_norm": 1.4543901681900024, "learning_rate": 7.606060606060607e-07, "num_tokens": 6928012.0, "completions/mean_length": 234.0, "completions/min_length": 228.0, "completions/max_length": 238.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 234.0, "completions/min_terminated_length": 228.0, "completions/max_terminated_length": 238.0, "rewards/meter/mean": 0.9989113807678223, "rewards/meter/std": 0.0005675117135979235, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989113807678223, "rewards/total_composite/std": 0.0005675117135979235, "reward": 0.9989113807678223, "reward_std": 0.0005675173015333712, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03521440923213959, "sampling/sampling_logp_difference/max": 1.4896221160888672, "sampling/importance_sampling_ratio/min": 0.22545784711837769, "sampling/importance_sampling_ratio/mean": 1.007856011390686, "sampling/importance_sampling_ratio/max": 1.7903379201889038, "entropy": 0.3369157761335373, "clip_ratio/low_mean": 0.004255396313965321, "clip_ratio/low_min": 0.004255396313965321, "clip_ratio/high_mean": 0.016509592533111572, "clip_ratio/high_max": 0.016509592533111572, "clip_ratio/region_mean": 0.020764988847076893, "reward_total_mean": 0.9989113807678223, "reward_meter_mean": 0.9989113807678223, "reward_meter_std": 0.0005675117135979235, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9989113807678223, "reward_total_composite_std": 0.0005675117135979235} {"timestamp_utc": "2026-04-12T03:15:15Z", "mode": "eval", "global_step": 3050, "epoch": 0.12250471944410973, "eval_loss": NaN, "eval_runtime": 77.5007, "eval_samples_per_second": 1.342, "eval_steps_per_second": 0.168, "eval_num_tokens": 6928012.0, "eval_completions/mean_length": 212.6346153846154, "eval_completions/min_length": 60.76923076923077, "eval_completions/max_length": 416.3076923076923, "eval_completions/clipped_ratio": 0.019230769230769232, "eval_completions/mean_terminated_length": 206.1826934814453, "eval_completions/min_terminated_length": 60.76923076923077, "eval_completions/max_terminated_length": 395.53846153846155, "eval_rewards/meter/mean": 0.7826263767022353, "eval_rewards/meter/std": 0.3530325018442594, "eval_rewards/count_adherence/mean": 0.9593305908716642, "eval_rewards/count_adherence/std": 0.06269322364376141, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/repeat_penalty/mean": 0.9508223029283377, "eval_rewards/repeat_penalty/std": 0.06783326672246823, "eval_rewards/total_composite/mean": 0.722142334167774, "eval_rewards/total_composite/std": 0.3449985155692467, "eval_reward": 0.722142334167774, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.030635818409231994, "eval_sampling/sampling_logp_difference/max": 1.1543501340425932, "eval_sampling/importance_sampling_ratio/min": 0.3223212957382202, "eval_sampling/importance_sampling_ratio/mean": 1.0082989380909846, "eval_sampling/importance_sampling_ratio/max": 1.5470959773430457, "eval_entropy": 0.34184010556110966, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.722142334167774, "eval_reward_meter_mean": 0.7826263767022353, "eval_reward_meter_std": 0.3530325018442594, "eval_reward_count_adherence_mean": 0.9593305908716642, "eval_reward_count_adherence_std": 0.06269322364376141, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_repeat_penalty_mean": 0.9508223029283377, "eval_reward_repeat_penalty_std": 0.06783326672246823, "eval_reward_total_composite_mean": 0.722142334167774, "eval_reward_total_composite_std": 0.3449985155692467} {"timestamp_utc": "2026-04-12T03:15:25Z", "mode": "train", "global_step": 3051, "epoch": 0.12254488492589469, "loss": 0.0034, "grad_norm": 1.5024135112762451, "learning_rate": 7.575757575757576e-07, "num_tokens": 6931580.0, "completions/mean_length": 233.0, "completions/min_length": 231.0, "completions/max_length": 235.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 233.0, "completions/min_terminated_length": 231.0, "completions/max_terminated_length": 235.0, "rewards/meter/mean": 0.999167799949646, "rewards/meter/std": 7.227841706480831e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9886363744735718, "rewards/repeat_penalty/std": 0.03214120864868164, "rewards/total_composite/mean": 0.9878134727478027, "rewards/total_composite/std": 0.03211057558655739, "reward": 0.9878134727478027, "reward_std": 0.03211057186126709, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.037654075771570206, "sampling/sampling_logp_difference/max": 1.2263059616088867, "sampling/importance_sampling_ratio/min": 0.29337432980537415, "sampling/importance_sampling_ratio/mean": 1.0094155073165894, "sampling/importance_sampling_ratio/max": 1.6730788946151733, "entropy": 0.3387974835932255, "clip_ratio/low_mean": 0.0026709402445703745, "clip_ratio/low_min": 0.0026709402445703745, "clip_ratio/high_mean": 0.02257009909953922, "clip_ratio/high_max": 0.02257009909953922, "clip_ratio/region_mean": 0.025241039344109595, "reward_total_mean": 0.9878134727478027, "reward_meter_mean": 0.999167799949646, "reward_meter_std": 7.227841706480831e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9886363744735718, "reward_repeat_penalty_std": 0.03214120864868164, "reward_total_composite_mean": 0.9878134727478027, "reward_total_composite_std": 0.03211057558655739} {"timestamp_utc": "2026-04-12T03:15:33Z", "mode": "train", "global_step": 3052, "epoch": 0.12258505040767964, "loss": 0.0461, "grad_norm": 2.0073533058166504, "learning_rate": 7.545454545454546e-07, "num_tokens": 6935794.0, "completions/mean_length": 322.75, "completions/min_length": 308.0, "completions/max_length": 356.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 322.75, "completions/min_terminated_length": 308.0, "completions/max_terminated_length": 356.0, "rewards/meter/mean": 0.9991399645805359, "rewards/meter/std": 0.00012341544788796455, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0578637570142746, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9916666746139526, "rewards/repeat_penalty/std": 0.0235702246427536, "rewards/total_composite/mean": 0.9595891237258911, "rewards/total_composite/std": 0.05744243040680885, "reward": 0.9595891237258911, "reward_std": 0.057442426681518555, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04169188067317009, "sampling/sampling_logp_difference/max": 1.2492733001708984, "sampling/importance_sampling_ratio/min": 0.2867130637168884, "sampling/importance_sampling_ratio/mean": 1.0115916728973389, "sampling/importance_sampling_ratio/max": 1.883813500404358, "entropy": 0.41726846620440483, "clip_ratio/low_mean": 0.009418385569006205, "clip_ratio/low_min": 0.009418385569006205, "clip_ratio/high_mean": 0.016292808344587684, "clip_ratio/high_max": 0.016292808344587684, "clip_ratio/region_mean": 0.02571119391359389, "reward_total_mean": 0.9595891237258911, "reward_meter_mean": 0.9991399645805359, "reward_meter_std": 0.00012341544788796455, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0578637570142746, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9916666746139526, "reward_repeat_penalty_std": 0.0235702246427536, "reward_total_composite_mean": 0.9595891237258911, "reward_total_composite_std": 0.05744243040680885} {"timestamp_utc": "2026-04-12T03:15:38Z", "mode": "train", "global_step": 3053, "epoch": 0.1226252158894646, "loss": 0.0049, "grad_norm": 2.259249448776245, "learning_rate": 7.515151515151516e-07, "num_tokens": 6937691.0, "completions/mean_length": 66.125, "completions/min_length": 65.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.125, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9928057193756104, "rewards/meter/std": 0.0009291421738453209, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9928057193756104, "rewards/total_composite/std": 0.0009291421738453209, "reward": 0.9928057193756104, "reward_std": 0.0009291406604461372, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015979096293449402, "sampling/sampling_logp_difference/max": 1.00278902053833, "sampling/importance_sampling_ratio/min": 0.36685484647750854, "sampling/importance_sampling_ratio/mean": 1.0060573816299438, "sampling/importance_sampling_ratio/max": 1.8138948678970337, "entropy": 0.10556314932182431, "clip_ratio/low_mean": 0.013010540511459112, "clip_ratio/low_min": 0.013010540511459112, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.013010540511459112, "reward_total_mean": 0.9928057193756104, "reward_meter_mean": 0.9928057193756104, "reward_meter_std": 0.0009291421738453209, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9928057193756104, "reward_total_composite_std": 0.0009291421738453209} {"timestamp_utc": "2026-04-12T03:15:44Z", "mode": "train", "global_step": 3054, "epoch": 0.12266538137124955, "loss": 0.0032, "grad_norm": 2.697394847869873, "learning_rate": 7.484848484848485e-07, "num_tokens": 6940138.0, "completions/mean_length": 137.875, "completions/min_length": 135.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 137.875, "completions/min_terminated_length": 135.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.9923188090324402, "rewards/meter/std": 0.004795641638338566, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9745643138885498, "rewards/total_composite/std": 0.04966309294104576, "reward": 0.9745643138885498, "reward_std": 0.04966310039162636, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02996024861931801, "sampling/sampling_logp_difference/max": 1.230236291885376, "sampling/importance_sampling_ratio/min": 0.2922235131263733, "sampling/importance_sampling_ratio/mean": 1.0037504434585571, "sampling/importance_sampling_ratio/max": 1.498881459236145, "entropy": 0.28184244222939014, "clip_ratio/low_mean": 0.0018115942366421223, "clip_ratio/low_min": 0.0018115942366421223, "clip_ratio/high_mean": 0.026310671470128, "clip_ratio/high_max": 0.026310671470128, "clip_ratio/region_mean": 0.028122265706770122, "reward_total_mean": 0.9745643138885498, "reward_meter_mean": 0.9923188090324402, "reward_meter_std": 0.004795641638338566, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.9745643138885498, "reward_total_composite_std": 0.04966309294104576} {"timestamp_utc": "2026-04-12T03:15:50Z", "mode": "train", "global_step": 3055, "epoch": 0.1227055468530345, "loss": 0.0153, "grad_norm": 2.5886197090148926, "learning_rate": 7.454545454545455e-07, "num_tokens": 6943559.0, "completions/mean_length": 233.625, "completions/min_length": 228.0, "completions/max_length": 238.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 233.625, "completions/min_terminated_length": 228.0, "completions/max_terminated_length": 238.0, "rewards/meter/mean": 0.9977272748947144, "rewards/meter/std": 0.0019398098811507225, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.932692289352417, "rewards/repeat_penalty/std": 0.08661473542451859, "rewards/total_composite/mean": 0.9304856061935425, "rewards/total_composite/std": 0.08541877567768097, "reward": 0.9304856061935425, "reward_std": 0.08541877567768097, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.045273296535015106, "sampling/sampling_logp_difference/max": 1.65692138671875, "sampling/importance_sampling_ratio/min": 0.19072523713111877, "sampling/importance_sampling_ratio/mean": 1.0053743124008179, "sampling/importance_sampling_ratio/max": 1.9301133155822754, "entropy": 0.3986949659883976, "clip_ratio/low_mean": 0.010666976682841778, "clip_ratio/low_min": 0.010666976682841778, "clip_ratio/high_mean": 0.025692227762192488, "clip_ratio/high_max": 0.025692227762192488, "clip_ratio/region_mean": 0.036359204445034266, "reward_total_mean": 0.9304856061935425, "reward_meter_mean": 0.9977272748947144, "reward_meter_std": 0.0019398098811507225, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.932692289352417, "reward_repeat_penalty_std": 0.08661473542451859, "reward_total_composite_mean": 0.9304856061935425, "reward_total_composite_std": 0.08541877567768097} {"timestamp_utc": "2026-04-12T03:15:55Z", "mode": "train", "global_step": 3056, "epoch": 0.12274571233481946, "loss": -0.0003, "grad_norm": 1.2885347604751587, "learning_rate": 7.424242424242425e-07, "num_tokens": 6946121.0, "completions/mean_length": 142.25, "completions/min_length": 138.0, "completions/max_length": 144.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 142.25, "completions/min_terminated_length": 138.0, "completions/max_terminated_length": 144.0, "rewards/meter/mean": 0.9991947412490845, "rewards/meter/std": 7.931316940812394e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9813523292541504, "rewards/total_composite/std": 0.050475094467401505, "reward": 0.9813523292541504, "reward_std": 0.05047507956624031, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015199076384305954, "sampling/sampling_logp_difference/max": 0.8146209716796875, "sampling/importance_sampling_ratio/min": 0.4428071081638336, "sampling/importance_sampling_ratio/mean": 1.004903793334961, "sampling/importance_sampling_ratio/max": 1.6997216939926147, "entropy": 0.18231415934860706, "clip_ratio/low_mean": 0.0017605633474886417, "clip_ratio/low_min": 0.0017605633474886417, "clip_ratio/high_mean": 0.01225746376439929, "clip_ratio/high_max": 0.01225746376439929, "clip_ratio/region_mean": 0.014018027111887932, "reward_total_mean": 0.9813523292541504, "reward_meter_mean": 0.9991947412490845, "reward_meter_std": 7.931316940812394e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.9813523292541504, "reward_total_composite_std": 0.050475094467401505} {"timestamp_utc": "2026-04-12T03:16:00Z", "mode": "train", "global_step": 3057, "epoch": 0.12278587781660441, "loss": -0.0002, "grad_norm": 0.10690882056951523, "learning_rate": 7.393939393939395e-07, "num_tokens": 6948039.0, "completions/mean_length": 66.75, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.75, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981523752212524, "rewards/meter/std": 3.7664287901861826e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981523752212524, "rewards/total_composite/std": 3.7664287901861826e-06, "reward": 0.9981523752212524, "reward_std": 3.7686852465412812e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006302142050117254, "sampling/sampling_logp_difference/max": 0.39133310317993164, "sampling/importance_sampling_ratio/min": 0.8453654646873474, "sampling/importance_sampling_ratio/mean": 1.0037543773651123, "sampling/importance_sampling_ratio/max": 1.4789509773254395, "entropy": 0.05063098296523094, "clip_ratio/low_mean": 0.0056535504991188645, "clip_ratio/low_min": 0.0056535504991188645, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0056535504991188645, "reward_total_mean": 0.9981523752212524, "reward_meter_mean": 0.9981523752212524, "reward_meter_std": 3.7664287901861826e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981523752212524, "reward_total_composite_std": 3.7664287901861826e-06} {"timestamp_utc": "2026-04-12T03:16:04Z", "mode": "train", "global_step": 3058, "epoch": 0.12282604329838936, "loss": 0.0021, "grad_norm": 2.2398717403411865, "learning_rate": 7.363636363636364e-07, "num_tokens": 6949869.0, "completions/mean_length": 65.75, "completions/min_length": 65.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.75, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9914587736129761, "rewards/meter/std": 0.00509608956053853, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9914587736129761, "rewards/total_composite/std": 0.00509608956053853, "reward": 0.9914587736129761, "reward_std": 0.005096097476780415, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011036393232643604, "sampling/sampling_logp_difference/max": 0.8546088933944702, "sampling/importance_sampling_ratio/min": 0.42544955015182495, "sampling/importance_sampling_ratio/mean": 1.003832221031189, "sampling/importance_sampling_ratio/max": 1.2646030187606812, "entropy": 0.07432558294385672, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/high_mean": 0.0038170163752511144, "clip_ratio/high_max": 0.0038170163752511144, "clip_ratio/region_mean": 0.005710955825634301, "reward_total_mean": 0.9914587736129761, "reward_meter_mean": 0.9914587736129761, "reward_meter_std": 0.00509608956053853, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9914587736129761, "reward_total_composite_std": 0.00509608956053853} {"timestamp_utc": "2026-04-12T03:16:10Z", "mode": "train", "global_step": 3059, "epoch": 0.12286620878017432, "loss": -0.0012, "grad_norm": 2.9161179065704346, "learning_rate": 7.333333333333334e-07, "num_tokens": 6952388.0, "completions/mean_length": 125.875, "completions/min_length": 123.0, "completions/max_length": 127.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 125.875, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.938283383846283, "rewards/meter/std": 0.15279191732406616, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9285714626312256, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.8751063346862793, "rewards/total_composite/std": 0.17424653470516205, "reward": 0.8751063346862793, "reward_std": 0.17424653470516205, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025627773255109787, "sampling/sampling_logp_difference/max": 0.9624452590942383, "sampling/importance_sampling_ratio/min": 0.3819577693939209, "sampling/importance_sampling_ratio/mean": 1.0073537826538086, "sampling/importance_sampling_ratio/max": 1.7322660684585571, "entropy": 0.24840066954493523, "clip_ratio/low_mean": 0.008024971117265522, "clip_ratio/low_min": 0.008024971117265522, "clip_ratio/high_mean": 0.006889763753861189, "clip_ratio/high_max": 0.006889763753861189, "clip_ratio/region_mean": 0.014914734871126711, "reward_total_mean": 0.8751063346862793, "reward_meter_mean": 0.938283383846283, "reward_meter_std": 0.15279191732406616, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9285714626312256, "reward_repeat_penalty_std": 0.07636035233736038, "reward_total_composite_mean": 0.8751063346862793, "reward_total_composite_std": 0.17424653470516205} {"timestamp_utc": "2026-04-12T03:16:15Z", "mode": "train", "global_step": 3060, "epoch": 0.12290637426195927, "loss": 0.0128, "grad_norm": 3.135732650756836, "learning_rate": 7.303030303030304e-07, "num_tokens": 6954749.0, "completions/mean_length": 138.125, "completions/min_length": 133.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 138.125, "completions/min_terminated_length": 133.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.9924278259277344, "rewards/meter/std": 0.0035289754159748554, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9924278259277344, "rewards/total_composite/std": 0.0035289754159748554, "reward": 0.9924278259277344, "reward_std": 0.0035289691295474768, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03332981467247009, "sampling/sampling_logp_difference/max": 0.9910516738891602, "sampling/importance_sampling_ratio/min": 0.37118610739707947, "sampling/importance_sampling_ratio/mean": 1.0079230070114136, "sampling/importance_sampling_ratio/max": 1.7354730367660522, "entropy": 0.32595139369368553, "clip_ratio/low_mean": 0.010748314904049039, "clip_ratio/low_min": 0.010748314904049039, "clip_ratio/high_mean": 0.01723209215560928, "clip_ratio/high_max": 0.01723209215560928, "clip_ratio/region_mean": 0.02798040705965832, "reward_total_mean": 0.9924278259277344, "reward_meter_mean": 0.9924278259277344, "reward_meter_std": 0.0035289754159748554, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9924278259277344, "reward_total_composite_std": 0.0035289754159748554} {"timestamp_utc": "2026-04-12T03:16:19Z", "mode": "train", "global_step": 3061, "epoch": 0.12294653974374423, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.272727272727273e-07, "num_tokens": 6956349.0, "completions/mean_length": 32.0, "completions/min_length": 32.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "reward": 0.9992982149124146, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 5.019194941269234e-05, "sampling/sampling_logp_difference/max": 0.0008984014857560396, "sampling/importance_sampling_ratio/min": 0.9998447299003601, "sampling/importance_sampling_ratio/mean": 1.0000483989715576, "sampling/importance_sampling_ratio/max": 1.000898838043213, "entropy": 0.000393166919820942, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9992982149124146, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:16:24Z", "mode": "train", "global_step": 3062, "epoch": 0.12298670522552918, "loss": 0.0008, "grad_norm": 1.910351037979126, "learning_rate": 7.242424242424243e-07, "num_tokens": 6958746.0, "completions/mean_length": 124.625, "completions/min_length": 123.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 124.625, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.997795820236206, "rewards/meter/std": 0.0004729351494461298, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.997795820236206, "rewards/total_composite/std": 0.0004729351494461298, "reward": 0.997795820236206, "reward_std": 0.0004729465872514993, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018115131184458733, "sampling/sampling_logp_difference/max": 1.9803028106689453, "sampling/importance_sampling_ratio/min": 0.13802744448184967, "sampling/importance_sampling_ratio/mean": 1.0041706562042236, "sampling/importance_sampling_ratio/max": 1.4670944213867188, "entropy": 0.1536051593720913, "clip_ratio/low_mean": 0.0010000000474974513, "clip_ratio/low_min": 0.0010000000474974513, "clip_ratio/high_mean": 0.01407297421246767, "clip_ratio/high_max": 0.01407297421246767, "clip_ratio/region_mean": 0.015072974259965122, "reward_total_mean": 0.997795820236206, "reward_meter_mean": 0.997795820236206, "reward_meter_std": 0.0004729351494461298, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.997795820236206, "reward_total_composite_std": 0.0004729351494461298} {"timestamp_utc": "2026-04-12T03:16:30Z", "mode": "train", "global_step": 3063, "epoch": 0.12302687070731413, "loss": 0.0173, "grad_norm": 3.4471962451934814, "learning_rate": 7.212121212121213e-07, "num_tokens": 6961175.0, "completions/mean_length": 137.625, "completions/min_length": 134.0, "completions/max_length": 143.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 137.625, "completions/min_terminated_length": 134.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.9925941228866577, "rewards/meter/std": 0.0024486577603965998, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9925941228866577, "rewards/total_composite/std": 0.0024486577603965998, "reward": 0.9925941228866577, "reward_std": 0.0024486577603965998, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04126367345452309, "sampling/sampling_logp_difference/max": 1.3208866119384766, "sampling/importance_sampling_ratio/min": 0.2668985426425934, "sampling/importance_sampling_ratio/mean": 1.0079936981201172, "sampling/importance_sampling_ratio/max": 1.8253576755523682, "entropy": 0.3563910871744156, "clip_ratio/low_mean": 0.014415563317015767, "clip_ratio/low_min": 0.014415563317015767, "clip_ratio/high_mean": 0.018338565016165376, "clip_ratio/high_max": 0.018338565016165376, "clip_ratio/region_mean": 0.03275412833318114, "reward_total_mean": 0.9925941228866577, "reward_meter_mean": 0.9925941228866577, "reward_meter_std": 0.0024486577603965998, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9925941228866577, "reward_total_composite_std": 0.0024486577603965998} {"timestamp_utc": "2026-04-12T03:16:34Z", "mode": "train", "global_step": 3064, "epoch": 0.12306703618909909, "loss": 0.0097, "grad_norm": 3.6814091205596924, "learning_rate": 7.181818181818182e-07, "num_tokens": 6963041.0, "completions/mean_length": 65.25, "completions/min_length": 63.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.25, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9920260310173035, "rewards/meter/std": 0.0032561845146119595, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9920260310173035, "rewards/total_composite/std": 0.0032561845146119595, "reward": 0.9920260310173035, "reward_std": 0.0032561977859586477, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031295765191316605, "sampling/sampling_logp_difference/max": 1.3808412551879883, "sampling/importance_sampling_ratio/min": 0.2513670027256012, "sampling/importance_sampling_ratio/mean": 1.0053472518920898, "sampling/importance_sampling_ratio/max": 1.6320992708206177, "entropy": 0.20045749936252832, "clip_ratio/low_mean": 0.009565269108861685, "clip_ratio/low_min": 0.009565269108861685, "clip_ratio/high_mean": 0.009527972200885415, "clip_ratio/high_max": 0.009527972200885415, "clip_ratio/region_mean": 0.0190932413097471, "reward_total_mean": 0.9920260310173035, "reward_meter_mean": 0.9920260310173035, "reward_meter_std": 0.0032561845146119595, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9920260310173035, "reward_total_composite_std": 0.0032561845146119595} {"timestamp_utc": "2026-04-12T03:16:42Z", "mode": "train", "global_step": 3065, "epoch": 0.12310720167088404, "loss": 0.0003, "grad_norm": 1.8719370365142822, "learning_rate": 7.151515151515153e-07, "num_tokens": 6967523.0, "completions/mean_length": 350.25, "completions/min_length": 343.0, "completions/max_length": 362.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 350.25, "completions/min_terminated_length": 343.0, "completions/max_terminated_length": 362.0, "rewards/meter/mean": 0.9957402944564819, "rewards/meter/std": 0.009437782689929008, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8961663246154785, "rewards/total_composite/std": 0.008493990637362003, "reward": 0.8961663246154785, "reward_std": 0.008493990637362003, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05361352488398552, "sampling/sampling_logp_difference/max": 2.0558176040649414, "sampling/importance_sampling_ratio/min": 0.12798814475536346, "sampling/importance_sampling_ratio/mean": 1.0094969272613525, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4838559664785862, "clip_ratio/low_mean": 0.004285714123398066, "clip_ratio/low_min": 0.004285714123398066, "clip_ratio/high_mean": 0.032739987364038825, "clip_ratio/high_max": 0.032739987364038825, "clip_ratio/region_mean": 0.03702570148743689, "reward_total_mean": 0.8961663246154785, "reward_meter_mean": 0.9957402944564819, "reward_meter_std": 0.009437782689929008, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8961663246154785, "reward_total_composite_std": 0.008493990637362003} {"timestamp_utc": "2026-04-12T03:16:46Z", "mode": "train", "global_step": 3066, "epoch": 0.123147367152669, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.121212121212122e-07, "num_tokens": 6969011.0, "completions/mean_length": 36.0, "completions/min_length": 36.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.0, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9996045231819153, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9996045231819153, "rewards/total_composite/std": 0.0, "reward": 0.9996045231819153, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0014014473417773843, "sampling/sampling_logp_difference/max": 0.034461501985788345, "sampling/importance_sampling_ratio/min": 0.9661255478858948, "sampling/importance_sampling_ratio/mean": 1.0010969638824463, "sampling/importance_sampling_ratio/max": 1.0258433818817139, "entropy": 0.013131682993844151, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9996045231819153, "reward_meter_mean": 0.9996045231819153, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9996045231819153, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:16:51Z", "mode": "train", "global_step": 3067, "epoch": 0.12318753263445395, "loss": -0.0038, "grad_norm": 3.1952340602874756, "learning_rate": 7.090909090909092e-07, "num_tokens": 6970813.0, "completions/mean_length": 69.25, "completions/min_length": 68.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.25, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9955199956893921, "rewards/meter/std": 0.0008645570487715304, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9955199956893921, "rewards/total_composite/std": 0.0008645570487715304, "reward": 0.9955199956893921, "reward_std": 0.0008645570487715304, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02916860766708851, "sampling/sampling_logp_difference/max": 1.2169954776763916, "sampling/importance_sampling_ratio/min": 0.29611852765083313, "sampling/importance_sampling_ratio/mean": 1.0018832683563232, "sampling/importance_sampling_ratio/max": 1.5354810953140259, "entropy": 0.21143894083797932, "clip_ratio/low_mean": 0.0036498295376077294, "clip_ratio/low_min": 0.0036498295376077294, "clip_ratio/high_mean": 0.010746001382358372, "clip_ratio/high_max": 0.010746001382358372, "clip_ratio/region_mean": 0.014395830919966102, "reward_total_mean": 0.9955199956893921, "reward_meter_mean": 0.9955199956893921, "reward_meter_std": 0.0008645570487715304, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9955199956893921, "reward_total_composite_std": 0.0008645570487715304} {"timestamp_utc": "2026-04-12T03:16:57Z", "mode": "train", "global_step": 3068, "epoch": 0.1232276981162389, "loss": 0.0041, "grad_norm": 1.5282530784606934, "learning_rate": 7.060606060606061e-07, "num_tokens": 6973664.0, "completions/mean_length": 177.375, "completions/min_length": 176.0, "completions/max_length": 179.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 177.375, "completions/min_terminated_length": 176.0, "completions/max_terminated_length": 179.0, "rewards/meter/mean": 0.9991249442100525, "rewards/meter/std": 0.00017947136075235903, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9861111044883728, "rewards/repeat_penalty/std": 0.03928370773792267, "rewards/total_composite/mean": 0.9852460026741028, "rewards/total_composite/std": 0.039191942662000656, "reward": 0.9852460026741028, "reward_std": 0.03919193148612976, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02057509496808052, "sampling/sampling_logp_difference/max": 1.0894832611083984, "sampling/importance_sampling_ratio/min": 0.33639028668403625, "sampling/importance_sampling_ratio/mean": 1.003198266029358, "sampling/importance_sampling_ratio/max": 1.8010437488555908, "entropy": 0.22719617374241352, "clip_ratio/low_mean": 0.0007022471982054412, "clip_ratio/low_min": 0.0007022471982054412, "clip_ratio/high_mean": 0.014811165689025074, "clip_ratio/high_max": 0.014811165689025074, "clip_ratio/region_mean": 0.015513412887230515, "reward_total_mean": 0.9852460026741028, "reward_meter_mean": 0.9991249442100525, "reward_meter_std": 0.00017947136075235903, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9861111044883728, "reward_repeat_penalty_std": 0.03928370773792267, "reward_total_composite_mean": 0.9852460026741028, "reward_total_composite_std": 0.039191942662000656} {"timestamp_utc": "2026-04-12T03:17:04Z", "mode": "train", "global_step": 3069, "epoch": 0.12326786359802386, "loss": 0.0046, "grad_norm": 1.41837477684021, "learning_rate": 7.03030303030303e-07, "num_tokens": 6977475.0, "completions/mean_length": 285.375, "completions/min_length": 282.0, "completions/max_length": 288.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 285.375, "completions/min_terminated_length": 282.0, "completions/max_terminated_length": 288.0, "rewards/meter/mean": 0.9989954233169556, "rewards/meter/std": 0.00014736292359884828, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9500000476837158, "rewards/repeat_penalty/std": 0.0471404492855072, "rewards/total_composite/mean": 0.9490443468093872, "rewards/total_composite/std": 0.04706759378314018, "reward": 0.9490443468093872, "reward_std": 0.04706757515668869, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026732422411441803, "sampling/sampling_logp_difference/max": 1.257065773010254, "sampling/importance_sampling_ratio/min": 0.2844875454902649, "sampling/importance_sampling_ratio/mean": 1.005640983581543, "sampling/importance_sampling_ratio/max": 1.655118703842163, "entropy": 0.3014299161732197, "clip_ratio/low_mean": 0.005277339863823727, "clip_ratio/low_min": 0.005277339863823727, "clip_ratio/high_mean": 0.007437994237989187, "clip_ratio/high_max": 0.007437994237989187, "clip_ratio/region_mean": 0.012715334101812914, "reward_total_mean": 0.9490443468093872, "reward_meter_mean": 0.9989954233169556, "reward_meter_std": 0.00014736292359884828, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9500000476837158, "reward_repeat_penalty_std": 0.0471404492855072, "reward_total_composite_mean": 0.9490443468093872, "reward_total_composite_std": 0.04706759378314018} {"timestamp_utc": "2026-04-12T03:17:10Z", "mode": "train", "global_step": 3070, "epoch": 0.12330802907980881, "loss": 0.0073, "grad_norm": 3.581768751144409, "learning_rate": 7.000000000000001e-07, "num_tokens": 6981013.0, "completions/mean_length": 234.25, "completions/min_length": 224.0, "completions/max_length": 250.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 234.25, "completions/min_terminated_length": 224.0, "completions/max_terminated_length": 250.0, "rewards/meter/mean": 0.9958624839782715, "rewards/meter/std": 0.004256864078342915, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9807692170143127, "rewards/repeat_penalty/std": 0.03560846298933029, "rewards/total_composite/mean": 0.9766829013824463, "rewards/total_composite/std": 0.034849490970373154, "reward": 0.9766829013824463, "reward_std": 0.03484949469566345, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05148389935493469, "sampling/sampling_logp_difference/max": 3.614051103591919, "sampling/importance_sampling_ratio/min": 0.026942478492856026, "sampling/importance_sampling_ratio/mean": 1.0083545446395874, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43540840223431587, "clip_ratio/low_mean": 0.005974947940558195, "clip_ratio/low_min": 0.005974947940558195, "clip_ratio/high_mean": 0.031633416656404734, "clip_ratio/high_max": 0.031633416656404734, "clip_ratio/region_mean": 0.03760836459696293, "reward_total_mean": 0.9766829013824463, "reward_meter_mean": 0.9958624839782715, "reward_meter_std": 0.004256864078342915, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9807692170143127, "reward_repeat_penalty_std": 0.03560846298933029, "reward_total_composite_mean": 0.9766829013824463, "reward_total_composite_std": 0.034849490970373154} {"timestamp_utc": "2026-04-12T03:17:15Z", "mode": "train", "global_step": 3071, "epoch": 0.12334819456159377, "loss": 0.0005, "grad_norm": 0.05822064355015755, "learning_rate": 6.969696969696971e-07, "num_tokens": 6982996.0, "completions/mean_length": 97.875, "completions/min_length": 97.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.875, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9994089603424072, "rewards/meter/std": 7.70060796639882e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994089603424072, "rewards/total_composite/std": 7.70060796639882e-06, "reward": 0.9994089603424072, "reward_std": 7.685384844080545e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004502264782786369, "sampling/sampling_logp_difference/max": 0.6515674591064453, "sampling/importance_sampling_ratio/min": 0.5212281346321106, "sampling/importance_sampling_ratio/mean": 1.0012834072113037, "sampling/importance_sampling_ratio/max": 1.2397805452346802, "entropy": 0.02978663402609527, "clip_ratio/low_mean": 0.0025510203558951616, "clip_ratio/low_min": 0.0025510203558951616, "clip_ratio/high_mean": 0.0025641699321568012, "clip_ratio/high_max": 0.0025641699321568012, "clip_ratio/region_mean": 0.005115190288051963, "reward_total_mean": 0.9994089603424072, "reward_meter_mean": 0.9994089603424072, "reward_meter_std": 7.70060796639882e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994089603424072, "reward_total_composite_std": 7.70060796639882e-06} {"timestamp_utc": "2026-04-12T03:17:19Z", "mode": "train", "global_step": 3072, "epoch": 0.12338836004337872, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.939393939393941e-07, "num_tokens": 6984748.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00010368959920015186, "sampling/sampling_logp_difference/max": 0.002008092822507024, "sampling/importance_sampling_ratio/min": 0.9984797239303589, "sampling/importance_sampling_ratio/mean": 1.000089168548584, "sampling/importance_sampling_ratio/max": 1.0020101070404053, "entropy": 0.0010117705096490681, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:17:24Z", "mode": "train", "global_step": 3073, "epoch": 0.12342852552516367, "loss": 0.0005, "grad_norm": 0.6883857846260071, "learning_rate": 6.90909090909091e-07, "num_tokens": 6986610.0, "completions/mean_length": 66.75, "completions/min_length": 66.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.75, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9981359243392944, "rewards/meter/std": 3.569047839846462e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981359243392944, "rewards/total_composite/std": 3.569047839846462e-05, "reward": 0.9981359243392944, "reward_std": 3.56822092726361e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013685759156942368, "sampling/sampling_logp_difference/max": 1.1063222885131836, "sampling/importance_sampling_ratio/min": 0.3307732045650482, "sampling/importance_sampling_ratio/mean": 0.9993820190429688, "sampling/importance_sampling_ratio/max": 1.4870829582214355, "entropy": 0.056920765433460474, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/high_mean": 0.007463517948053777, "clip_ratio/high_max": 0.007463517948053777, "clip_ratio/region_mean": 0.009357457398436964, "reward_total_mean": 0.9981359243392944, "reward_meter_mean": 0.9981359243392944, "reward_meter_std": 3.569047839846462e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981359243392944, "reward_total_composite_std": 3.569047839846462e-05} {"timestamp_utc": "2026-04-12T03:17:29Z", "mode": "train", "global_step": 3074, "epoch": 0.12346869100694863, "loss": 0.0032, "grad_norm": 1.0535986423492432, "learning_rate": 6.878787878787879e-07, "num_tokens": 6989111.0, "completions/mean_length": 142.625, "completions/min_length": 140.0, "completions/max_length": 144.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 142.625, "completions/min_terminated_length": 140.0, "completions/max_terminated_length": 144.0, "rewards/meter/mean": 0.9991422891616821, "rewards/meter/std": 0.0001949365541804582, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9813013076782227, "rewards/total_composite/std": 0.05048074945807457, "reward": 0.9813013076782227, "reward_std": 0.050480738282203674, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018873082473874092, "sampling/sampling_logp_difference/max": 1.6918869018554688, "sampling/importance_sampling_ratio/min": 0.18417169153690338, "sampling/importance_sampling_ratio/mean": 1.0043222904205322, "sampling/importance_sampling_ratio/max": 1.7058764696121216, "entropy": 0.18659264594316483, "clip_ratio/low_mean": 0.003496503457427025, "clip_ratio/low_min": 0.003496503457427025, "clip_ratio/high_mean": 0.009702706767711788, "clip_ratio/high_max": 0.009702706767711788, "clip_ratio/region_mean": 0.013199210225138813, "reward_total_mean": 0.9813013076782227, "reward_meter_mean": 0.9991422891616821, "reward_meter_std": 0.0001949365541804582, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.9813013076782227, "reward_total_composite_std": 0.05048074945807457} {"timestamp_utc": "2026-04-12T03:17:34Z", "mode": "train", "global_step": 3075, "epoch": 0.12350885648873358, "loss": 0.0244, "grad_norm": 3.619894504547119, "learning_rate": 6.848484848484849e-07, "num_tokens": 6990969.0, "completions/mean_length": 69.25, "completions/min_length": 66.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.25, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9823685884475708, "rewards/meter/std": 0.03472607955336571, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9823685884475708, "rewards/total_composite/std": 0.03472607955336571, "reward": 0.9823685884475708, "reward_std": 0.034726083278656006, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030778199434280396, "sampling/sampling_logp_difference/max": 1.412278175354004, "sampling/importance_sampling_ratio/min": 0.24358773231506348, "sampling/importance_sampling_ratio/mean": 1.006119728088379, "sampling/importance_sampling_ratio/max": 1.8515444993972778, "entropy": 0.24566587805747986, "clip_ratio/low_mean": 0.0034246575087308884, "clip_ratio/low_min": 0.0034246575087308884, "clip_ratio/high_mean": 0.021908456226810813, "clip_ratio/high_max": 0.021908456226810813, "clip_ratio/region_mean": 0.0253331137355417, "reward_total_mean": 0.9823685884475708, "reward_meter_mean": 0.9823685884475708, "reward_meter_std": 0.03472607955336571, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9823685884475708, "reward_total_composite_std": 0.03472607955336571} {"timestamp_utc": "2026-04-12T03:17:38Z", "mode": "train", "global_step": 3076, "epoch": 0.12354902197051854, "loss": 0.0003, "grad_norm": 0.014152090065181255, "learning_rate": 6.818181818181818e-07, "num_tokens": 6992713.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.997339129447937, "rewards/meter/std": 3.875939285080676e-07, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.997339129447937, "rewards/total_composite/std": 3.875939285080676e-07, "reward": 0.997339129447937, "reward_std": 3.8628223819614504e-07, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004271084442734718, "sampling/sampling_logp_difference/max": 0.5116375088691711, "sampling/importance_sampling_ratio/min": 0.5995131134986877, "sampling/importance_sampling_ratio/mean": 0.9999513030052185, "sampling/importance_sampling_ratio/max": 1.2609317302703857, "entropy": 0.02187307458370924, "clip_ratio/low_mean": 0.004098360426723957, "clip_ratio/low_min": 0.004098360426723957, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.004098360426723957, "reward_total_mean": 0.997339129447937, "reward_meter_mean": 0.997339129447937, "reward_meter_std": 3.875939285080676e-07, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.997339129447937, "reward_total_composite_std": 3.875939285080676e-07} {"timestamp_utc": "2026-04-12T03:17:43Z", "mode": "train", "global_step": 3077, "epoch": 0.12358918745230349, "loss": 0.0002, "grad_norm": 0.27777397632598877, "learning_rate": 6.78787878787879e-07, "num_tokens": 6994530.0, "completions/mean_length": 68.125, "completions/min_length": 68.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.125, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9994935989379883, "rewards/meter/std": 8.932632226787973e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994935989379883, "rewards/total_composite/std": 8.932632226787973e-06, "reward": 0.9994935989379883, "reward_std": 8.932575838116463e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00869977567344904, "sampling/sampling_logp_difference/max": 1.2481575012207031, "sampling/importance_sampling_ratio/min": 0.28703317046165466, "sampling/importance_sampling_ratio/mean": 1.0012462139129639, "sampling/importance_sampling_ratio/max": 1.1400055885314941, "entropy": 0.050312931183725595, "clip_ratio/low_mean": 0.0018382353009656072, "clip_ratio/low_min": 0.0018382353009656072, "clip_ratio/high_mean": 0.0018115942366421223, "clip_ratio/high_max": 0.0018115942366421223, "clip_ratio/region_mean": 0.0036498295376077294, "reward_total_mean": 0.9994935989379883, "reward_meter_mean": 0.9994935989379883, "reward_meter_std": 8.932632226787973e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994935989379883, "reward_total_composite_std": 8.932632226787973e-06} {"timestamp_utc": "2026-04-12T03:17:48Z", "mode": "train", "global_step": 3078, "epoch": 0.12362935293408844, "loss": -0.0002, "grad_norm": 0.11041238158941269, "learning_rate": 6.757575757575759e-07, "num_tokens": 6996666.0, "completions/mean_length": 98.0, "completions/min_length": 98.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.0, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9994069337844849, "rewards/meter/std": 4.966230790159898e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994069337844849, "rewards/total_composite/std": 4.966230790159898e-06, "reward": 0.9994069337844849, "reward_std": 4.970342160959262e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003691497491672635, "sampling/sampling_logp_difference/max": 0.6290616989135742, "sampling/importance_sampling_ratio/min": 0.5330917835235596, "sampling/importance_sampling_ratio/mean": 0.9998006820678711, "sampling/importance_sampling_ratio/max": 1.217292070388794, "entropy": 0.032520961947739124, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9994069337844849, "reward_meter_mean": 0.9994069337844849, "reward_meter_std": 4.966230790159898e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994069337844849, "reward_total_composite_std": 4.966230790159898e-06} {"timestamp_utc": "2026-04-12T03:17:54Z", "mode": "train", "global_step": 3079, "epoch": 0.1236695184158734, "loss": 0.0004, "grad_norm": 1.676236867904663, "learning_rate": 6.727272727272728e-07, "num_tokens": 6999685.0, "completions/mean_length": 178.375, "completions/min_length": 177.0, "completions/max_length": 179.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 178.375, "completions/min_terminated_length": 177.0, "completions/max_terminated_length": 179.0, "rewards/meter/mean": 0.9991593360900879, "rewards/meter/std": 8.344750676769763e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333134651184, "rewards/repeat_penalty/std": 0.08266931027173996, "rewards/total_composite/mean": 0.9575244188308716, "rewards/total_composite/std": 0.08255957812070847, "reward": 0.9575244188308716, "reward_std": 0.08255956321954727, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020297784358263016, "sampling/sampling_logp_difference/max": 1.107102394104004, "sampling/importance_sampling_ratio/min": 0.3305152654647827, "sampling/importance_sampling_ratio/mean": 1.0064208507537842, "sampling/importance_sampling_ratio/max": 1.5720078945159912, "entropy": 0.23156088776886463, "clip_ratio/low_mean": 0.004197834758087993, "clip_ratio/low_min": 0.004197834758087993, "clip_ratio/high_mean": 0.009800075204111636, "clip_ratio/high_max": 0.009800075204111636, "clip_ratio/region_mean": 0.013997909962199628, "reward_total_mean": 0.9575244188308716, "reward_meter_mean": 0.9991593360900879, "reward_meter_std": 8.344750676769763e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333134651184, "reward_repeat_penalty_std": 0.08266931027173996, "reward_total_composite_mean": 0.9575244188308716, "reward_total_composite_std": 0.08255957812070847} {"timestamp_utc": "2026-04-12T03:17:58Z", "mode": "train", "global_step": 3080, "epoch": 0.12370968389765835, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.696969696969698e-07, "num_tokens": 7001620.0, "completions/mean_length": 79.875, "completions/min_length": 79.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.875, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "reward": 0.7190229296684265, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0004583710106089711, "sampling/sampling_logp_difference/max": 0.014358794316649437, "sampling/importance_sampling_ratio/min": 0.9983435869216919, "sampling/importance_sampling_ratio/mean": 1.0004509687423706, "sampling/importance_sampling_ratio/max": 1.0144623517990112, "entropy": 0.0046348033065442, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7190229296684265, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:18:03Z", "mode": "train", "global_step": 3081, "epoch": 0.1237498493794433, "loss": 0.0079, "grad_norm": 2.4082579612731934, "learning_rate": 6.666666666666667e-07, "num_tokens": 7004021.0, "completions/mean_length": 126.125, "completions/min_length": 124.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 126.125, "completions/min_terminated_length": 124.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9883460402488708, "rewards/meter/std": 0.006720090284943581, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9353639483451843, "rewards/total_composite/std": 0.07287098467350006, "reward": 0.9353639483451843, "reward_std": 0.07287097722291946, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03183456510305405, "sampling/sampling_logp_difference/max": 1.410654067993164, "sampling/importance_sampling_ratio/min": 0.243983656167984, "sampling/importance_sampling_ratio/mean": 1.0118496417999268, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36210910603404045, "clip_ratio/low_mean": 0.005876097013242543, "clip_ratio/low_min": 0.005876097013242543, "clip_ratio/high_mean": 0.02169638266786933, "clip_ratio/high_max": 0.02169638266786933, "clip_ratio/region_mean": 0.027572479681111872, "reward_total_mean": 0.9353639483451843, "reward_meter_mean": 0.9883460402488708, "reward_meter_std": 0.006720090284943581, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_total_composite_mean": 0.9353639483451843, "reward_total_composite_std": 0.07287098467350006} {"timestamp_utc": "2026-04-12T03:18:11Z", "mode": "train", "global_step": 3082, "epoch": 0.12379001486122826, "loss": 0.0066, "grad_norm": 3.0218822956085205, "learning_rate": 6.636363636363636e-07, "num_tokens": 7008316.0, "completions/mean_length": 331.875, "completions/min_length": 316.0, "completions/max_length": 340.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 331.875, "completions/min_terminated_length": 316.0, "completions/max_terminated_length": 340.0, "rewards/meter/mean": 0.9956144094467163, "rewards/meter/std": 0.0046219234354794025, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9868420958518982, "rewards/repeat_penalty/std": 0.024363677948713303, "rewards/total_composite/mean": 0.8187316656112671, "rewards/total_composite/std": 0.019191961735486984, "reward": 0.8187316656112671, "reward_std": 0.019191952422261238, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06327174603939056, "sampling/sampling_logp_difference/max": 4.011778354644775, "sampling/importance_sampling_ratio/min": 0.018101176247000694, "sampling/importance_sampling_ratio/mean": 1.016357183456421, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.594636969268322, "clip_ratio/low_mean": 0.011420327704399824, "clip_ratio/low_min": 0.011420327704399824, "clip_ratio/high_mean": 0.03181355516426265, "clip_ratio/high_max": 0.03181355516426265, "clip_ratio/region_mean": 0.04323388286866248, "reward_total_mean": 0.8187316656112671, "reward_meter_mean": 0.9956144094467163, "reward_meter_std": 0.0046219234354794025, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9868420958518982, "reward_repeat_penalty_std": 0.024363677948713303, "reward_total_composite_mean": 0.8187316656112671, "reward_total_composite_std": 0.019191961735486984} {"timestamp_utc": "2026-04-12T03:18:16Z", "mode": "train", "global_step": 3083, "epoch": 0.12383018034301321, "loss": -0.001, "grad_norm": 3.886202335357666, "learning_rate": 6.606060606060606e-07, "num_tokens": 7010295.0, "completions/mean_length": 68.375, "completions/min_length": 68.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.375, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9993771314620972, "rewards/meter/std": 0.00016229395987465978, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993771314620972, "rewards/total_composite/std": 0.00016229395987465978, "reward": 0.9993771314620972, "reward_std": 0.00016229994071181864, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0466373972594738, "sampling/sampling_logp_difference/max": 3.3516950607299805, "sampling/importance_sampling_ratio/min": 0.03502493351697922, "sampling/importance_sampling_ratio/mean": 0.9933637380599976, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12113892659544945, "clip_ratio/low_mean": 0.0018382353009656072, "clip_ratio/low_min": 0.0018382353009656072, "clip_ratio/high_mean": 0.023763853823766112, "clip_ratio/high_max": 0.023763853823766112, "clip_ratio/region_mean": 0.02560208912473172, "reward_total_mean": 0.9993771314620972, "reward_meter_mean": 0.9993771314620972, "reward_meter_std": 0.00016229395987465978, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993771314620972, "reward_total_composite_std": 0.00016229395987465978} {"timestamp_utc": "2026-04-12T03:18:21Z", "mode": "train", "global_step": 3084, "epoch": 0.12387034582479817, "loss": -0.0038, "grad_norm": 7.438833713531494, "learning_rate": 6.575757575757575e-07, "num_tokens": 7012093.0, "completions/mean_length": 70.75, "completions/min_length": 69.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.75, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9911935925483704, "rewards/meter/std": 0.011583052575588226, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9911935925483704, "rewards/total_composite/std": 0.011583052575588226, "reward": 0.9911935925483704, "reward_std": 0.011583039537072182, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.034917399287223816, "sampling/sampling_logp_difference/max": 1.0490436553955078, "sampling/importance_sampling_ratio/min": 0.3502725660800934, "sampling/importance_sampling_ratio/mean": 0.999201238155365, "sampling/importance_sampling_ratio/max": 1.5489575862884521, "entropy": 0.2923265937715769, "clip_ratio/low_mean": 0.0036231884732842445, "clip_ratio/low_min": 0.0036231884732842445, "clip_ratio/high_mean": 0.03329520847182721, "clip_ratio/high_max": 0.03329520847182721, "clip_ratio/region_mean": 0.036918396945111454, "reward_total_mean": 0.9911935925483704, "reward_meter_mean": 0.9911935925483704, "reward_meter_std": 0.011583052575588226, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9911935925483704, "reward_total_composite_std": 0.011583052575588226} {"timestamp_utc": "2026-04-12T03:18:27Z", "mode": "train", "global_step": 3085, "epoch": 0.12391051130658312, "loss": 0.0045, "grad_norm": 1.1208101511001587, "learning_rate": 6.545454545454547e-07, "num_tokens": 7014791.0, "completions/mean_length": 156.25, "completions/min_length": 150.0, "completions/max_length": 159.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 156.25, "completions/min_terminated_length": 150.0, "completions/max_terminated_length": 159.0, "rewards/meter/mean": 0.9991309642791748, "rewards/meter/std": 0.00010435016884002835, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991309642791748, "rewards/total_composite/std": 0.00010435016884002835, "reward": 0.9991309642791748, "reward_std": 0.00010435186413815245, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03277255594730377, "sampling/sampling_logp_difference/max": 0.7845733165740967, "sampling/importance_sampling_ratio/min": 0.456314355134964, "sampling/importance_sampling_ratio/mean": 1.0100115537643433, "sampling/importance_sampling_ratio/max": 1.801207184791565, "entropy": 0.3303154893219471, "clip_ratio/low_mean": 0.006369811948388815, "clip_ratio/low_min": 0.006369811948388815, "clip_ratio/high_mean": 0.01920302864164114, "clip_ratio/high_max": 0.01920302864164114, "clip_ratio/region_mean": 0.025572840590029955, "reward_total_mean": 0.9991309642791748, "reward_meter_mean": 0.9991309642791748, "reward_meter_std": 0.00010435016884002835, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9991309642791748, "reward_total_composite_std": 0.00010435016884002835} {"timestamp_utc": "2026-04-12T03:18:32Z", "mode": "train", "global_step": 3086, "epoch": 0.12395067678836807, "loss": 0.0065, "grad_norm": 4.402876377105713, "learning_rate": 6.515151515151516e-07, "num_tokens": 7017132.0, "completions/mean_length": 133.625, "completions/min_length": 132.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 133.625, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.998715341091156, "rewards/meter/std": 0.0006201790529303253, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998715341091156, "rewards/total_composite/std": 0.0006201790529303253, "reward": 0.998715341091156, "reward_std": 0.00062018126482144, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03640494868159294, "sampling/sampling_logp_difference/max": 1.8164355754852295, "sampling/importance_sampling_ratio/min": 0.16260430216789246, "sampling/importance_sampling_ratio/mean": 1.0022952556610107, "sampling/importance_sampling_ratio/max": 1.980418086051941, "entropy": 0.2722382918000221, "clip_ratio/low_mean": 0.010255418019369245, "clip_ratio/low_min": 0.010255418019369245, "clip_ratio/high_mean": 0.028133108280599117, "clip_ratio/high_max": 0.028133108280599117, "clip_ratio/region_mean": 0.03838852629996836, "reward_total_mean": 0.998715341091156, "reward_meter_mean": 0.998715341091156, "reward_meter_std": 0.0006201790529303253, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.998715341091156, "reward_total_composite_std": 0.0006201790529303253} {"timestamp_utc": "2026-04-12T03:18:37Z", "mode": "train", "global_step": 3087, "epoch": 0.12399084227015303, "loss": -0.0, "grad_norm": 0.165548175573349, "learning_rate": 6.484848484848485e-07, "num_tokens": 7018852.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9973361492156982, "rewards/meter/std": 8.160675861290656e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973361492156982, "rewards/total_composite/std": 8.160675861290656e-06, "reward": 0.9973361492156982, "reward_std": 8.157768206729088e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003540643723681569, "sampling/sampling_logp_difference/max": 0.9303553104400635, "sampling/importance_sampling_ratio/min": 0.394413560628891, "sampling/importance_sampling_ratio/mean": 0.9994197487831116, "sampling/importance_sampling_ratio/max": 1.119980812072754, "entropy": 0.016273654997348785, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0020491802133619785, "clip_ratio/high_max": 0.0020491802133619785, "clip_ratio/region_mean": 0.0020491802133619785, "reward_total_mean": 0.9973361492156982, "reward_meter_mean": 0.9973361492156982, "reward_meter_std": 8.160675861290656e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9973361492156982, "reward_total_composite_std": 8.160675861290656e-06} {"timestamp_utc": "2026-04-12T03:18:42Z", "mode": "train", "global_step": 3088, "epoch": 0.12403100775193798, "loss": 0.0063, "grad_norm": 5.205334186553955, "learning_rate": 6.454545454545455e-07, "num_tokens": 7021320.0, "completions/mean_length": 123.5, "completions/min_length": 121.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 123.5, "completions/min_terminated_length": 121.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.9974862337112427, "rewards/meter/std": 0.00045686395606026053, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9796685576438904, "rewards/total_composite/std": 0.05027516558766365, "reward": 0.9796685576438904, "reward_std": 0.050275154411792755, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0200046356767416, "sampling/sampling_logp_difference/max": 1.3565974235534668, "sampling/importance_sampling_ratio/min": 0.25753557682037354, "sampling/importance_sampling_ratio/mean": 1.004548192024231, "sampling/importance_sampling_ratio/max": 1.8565744161605835, "entropy": 0.1684292033314705, "clip_ratio/low_mean": 0.0020000000949949026, "clip_ratio/low_min": 0.0020000000949949026, "clip_ratio/high_mean": 0.014212732203304768, "clip_ratio/high_max": 0.014212732203304768, "clip_ratio/region_mean": 0.01621273229829967, "reward_total_mean": 0.9796685576438904, "reward_meter_mean": 0.9974862337112427, "reward_meter_std": 0.00045686395606026053, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.9796685576438904, "reward_total_composite_std": 0.05027516558766365} {"timestamp_utc": "2026-04-12T03:18:50Z", "mode": "train", "global_step": 3089, "epoch": 0.12407117323372294, "loss": -0.0405, "grad_norm": 1.6393530368804932, "learning_rate": 6.424242424242424e-07, "num_tokens": 7025761.0, "completions/mean_length": 360.125, "completions/min_length": 344.0, "completions/max_length": 397.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 360.125, "completions/min_terminated_length": 344.0, "completions/max_terminated_length": 397.0, "rewards/meter/mean": 0.9991971254348755, "rewards/meter/std": 0.00017116445815190673, "rewards/count_adherence/mean": 0.8409091234207153, "rewards/count_adherence/std": 0.04208274558186531, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9860681295394897, "rewards/repeat_penalty/std": 0.02584986388683319, "rewards/total_composite/mean": 0.8282498717308044, "rewards/total_composite/std": 0.040605854243040085, "reward": 0.8282498717308044, "reward_std": 0.04060585796833038, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04586139693856239, "sampling/sampling_logp_difference/max": 1.3152236938476562, "sampling/importance_sampling_ratio/min": 0.2684142589569092, "sampling/importance_sampling_ratio/mean": 1.0105684995651245, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4755103662610054, "clip_ratio/low_mean": 0.01749891194049269, "clip_ratio/low_min": 0.01749891194049269, "clip_ratio/high_mean": 0.007637429982423782, "clip_ratio/high_max": 0.007637429982423782, "clip_ratio/region_mean": 0.025136341922916472, "reward_total_mean": 0.8282498717308044, "reward_meter_mean": 0.9991971254348755, "reward_meter_std": 0.00017116445815190673, "reward_count_adherence_mean": 0.8409091234207153, "reward_count_adherence_std": 0.04208274558186531, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9860681295394897, "reward_repeat_penalty_std": 0.02584986388683319, "reward_total_composite_mean": 0.8282498717308044, "reward_total_composite_std": 0.040605854243040085} {"timestamp_utc": "2026-04-12T03:18:55Z", "mode": "train", "global_step": 3090, "epoch": 0.12411133871550789, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.393939393939394e-07, "num_tokens": 7027640.0, "completions/mean_length": 79.875, "completions/min_length": 79.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.875, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "reward": 0.7190229296684265, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.001453942502848804, "sampling/sampling_logp_difference/max": 0.6242718696594238, "sampling/importance_sampling_ratio/min": 0.5356513261795044, "sampling/importance_sampling_ratio/mean": 0.9997469782829285, "sampling/importance_sampling_ratio/max": 1.015737533569336, "entropy": 0.004133307666052133, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7190229296684265, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:18:59Z", "mode": "train", "global_step": 3091, "epoch": 0.12415150419729284, "loss": 0.0015, "grad_norm": 0.5760459899902344, "learning_rate": 6.363636363636364e-07, "num_tokens": 7029612.0, "completions/mean_length": 71.5, "completions/min_length": 70.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.5, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9994015097618103, "rewards/meter/std": 4.198703754809685e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994015097618103, "rewards/total_composite/std": 4.198703754809685e-05, "reward": 0.9994015097618103, "reward_std": 4.200140756438486e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010645559057593346, "sampling/sampling_logp_difference/max": 0.8983035087585449, "sampling/importance_sampling_ratio/min": 0.4072599709033966, "sampling/importance_sampling_ratio/mean": 1.0001534223556519, "sampling/importance_sampling_ratio/max": 1.4178498983383179, "entropy": 0.09367726277559996, "clip_ratio/low_mean": 0.0034722222480922937, "clip_ratio/low_min": 0.0034722222480922937, "clip_ratio/high_mean": 0.007018499891273677, "clip_ratio/high_max": 0.007018499891273677, "clip_ratio/region_mean": 0.010490722139365971, "reward_total_mean": 0.9994015097618103, "reward_meter_mean": 0.9994015097618103, "reward_meter_std": 4.198703754809685e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994015097618103, "reward_total_composite_std": 4.198703754809685e-05} {"timestamp_utc": "2026-04-12T03:19:05Z", "mode": "train", "global_step": 3092, "epoch": 0.1241916696790778, "loss": -0.0027, "grad_norm": 2.74900484085083, "learning_rate": 6.333333333333334e-07, "num_tokens": 7031807.0, "completions/mean_length": 117.375, "completions/min_length": 116.0, "completions/max_length": 120.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 117.375, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.9986400604248047, "rewards/meter/std": 0.0013286188477650285, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9737505912780762, "rewards/total_composite/std": 0.07163625955581665, "reward": 0.9737505912780762, "reward_std": 0.07163627445697784, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03260769695043564, "sampling/sampling_logp_difference/max": 1.1829195022583008, "sampling/importance_sampling_ratio/min": 0.306382954120636, "sampling/importance_sampling_ratio/mean": 1.0021259784698486, "sampling/importance_sampling_ratio/max": 1.3800021409988403, "entropy": 0.2854463141411543, "clip_ratio/low_mean": 0.0032327587250620127, "clip_ratio/low_min": 0.0032327587250620127, "clip_ratio/high_mean": 0.01596410130150616, "clip_ratio/high_max": 0.01596410130150616, "clip_ratio/region_mean": 0.019196860026568174, "reward_total_mean": 0.9737505912780762, "reward_meter_mean": 0.9986400604248047, "reward_meter_std": 0.0013286188477650285, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9737505912780762, "reward_total_composite_std": 0.07163625955581665} {"timestamp_utc": "2026-04-12T03:19:09Z", "mode": "train", "global_step": 3093, "epoch": 0.12423183516086275, "loss": 0.0001, "grad_norm": 0.078668013215065, "learning_rate": 6.303030303030304e-07, "num_tokens": 7033567.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9973331689834595, "rewards/meter/std": 1.0622773515933659e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973331689834595, "rewards/total_composite/std": 1.0622773515933659e-05, "reward": 0.9973331689834595, "reward_std": 1.0613573067530524e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0038269164506345987, "sampling/sampling_logp_difference/max": 0.5654642581939697, "sampling/importance_sampling_ratio/min": 0.7023022770881653, "sampling/importance_sampling_ratio/mean": 1.0025733709335327, "sampling/importance_sampling_ratio/max": 1.7602647542953491, "entropy": 0.020028789876960218, "clip_ratio/low_mean": 0.004098360426723957, "clip_ratio/low_min": 0.004098360426723957, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.004098360426723957, "reward_total_mean": 0.9973331689834595, "reward_meter_mean": 0.9973331689834595, "reward_meter_std": 1.0622773515933659e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9973331689834595, "reward_total_composite_std": 1.0622773515933659e-05} {"timestamp_utc": "2026-04-12T03:19:14Z", "mode": "train", "global_step": 3094, "epoch": 0.1242720006426477, "loss": 0.0003, "grad_norm": 0.07660355418920517, "learning_rate": 6.272727272727273e-07, "num_tokens": 7035327.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9973366260528564, "rewards/meter/std": 8.457681360596325e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973366260528564, "rewards/total_composite/std": 8.457681360596325e-06, "reward": 0.9973366260528564, "reward_std": 8.46226976136677e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008617924526333809, "sampling/sampling_logp_difference/max": 1.087322473526001, "sampling/importance_sampling_ratio/min": 0.3371179401874542, "sampling/importance_sampling_ratio/mean": 0.9982660412788391, "sampling/importance_sampling_ratio/max": 1.83155357837677, "entropy": 0.024718819418922067, "clip_ratio/low_mean": 0.0020491802133619785, "clip_ratio/low_min": 0.0020491802133619785, "clip_ratio/high_mean": 0.010245901066809893, "clip_ratio/high_max": 0.010245901066809893, "clip_ratio/region_mean": 0.012295081280171871, "reward_total_mean": 0.9973366260528564, "reward_meter_mean": 0.9973366260528564, "reward_meter_std": 8.457681360596325e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9973366260528564, "reward_total_composite_std": 8.457681360596325e-06} {"timestamp_utc": "2026-04-12T03:19:19Z", "mode": "train", "global_step": 3095, "epoch": 0.12431216612443266, "loss": 0.0003, "grad_norm": 0.8752101063728333, "learning_rate": 6.242424242424243e-07, "num_tokens": 7037833.0, "completions/mean_length": 131.25, "completions/min_length": 130.0, "completions/max_length": 132.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.25, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.9993793964385986, "rewards/meter/std": 9.467204654356465e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993793964385986, "rewards/total_composite/std": 9.467204654356465e-05, "reward": 0.9993793964385986, "reward_std": 9.466917254030704e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008658221922814846, "sampling/sampling_logp_difference/max": 0.8032641410827637, "sampling/importance_sampling_ratio/min": 0.44786468148231506, "sampling/importance_sampling_ratio/mean": 1.0029832124710083, "sampling/importance_sampling_ratio/max": 1.359288215637207, "entropy": 0.08324715122580528, "clip_ratio/low_mean": 0.0028625954291783273, "clip_ratio/low_min": 0.0028625954291783273, "clip_ratio/high_mean": 0.0019085081876255572, "clip_ratio/high_max": 0.0019085081876255572, "clip_ratio/region_mean": 0.0047711036168038845, "reward_total_mean": 0.9993793964385986, "reward_meter_mean": 0.9993793964385986, "reward_meter_std": 9.467204654356465e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993793964385986, "reward_total_composite_std": 9.467204654356465e-05} {"timestamp_utc": "2026-04-12T03:19:24Z", "mode": "train", "global_step": 3096, "epoch": 0.12435233160621761, "loss": -0.0004, "grad_norm": 0.717937171459198, "learning_rate": 6.212121212121212e-07, "num_tokens": 7039952.0, "completions/mean_length": 97.875, "completions/min_length": 97.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.875, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9993786811828613, "rewards/meter/std": 8.659059676574543e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993786811828613, "rewards/total_composite/std": 8.659059676574543e-05, "reward": 0.9993786811828613, "reward_std": 8.658833394292742e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006720404606312513, "sampling/sampling_logp_difference/max": 0.9676318168640137, "sampling/importance_sampling_ratio/min": 0.3799818456172943, "sampling/importance_sampling_ratio/mean": 0.9991164803504944, "sampling/importance_sampling_ratio/max": 1.574755072593689, "entropy": 0.03362013725563884, "clip_ratio/low_mean": 0.0025510203558951616, "clip_ratio/low_min": 0.0025510203558951616, "clip_ratio/high_mean": 0.006390700465999544, "clip_ratio/high_max": 0.006390700465999544, "clip_ratio/region_mean": 0.008941720821894705, "reward_total_mean": 0.9993786811828613, "reward_meter_mean": 0.9993786811828613, "reward_meter_std": 8.659059676574543e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993786811828613, "reward_total_composite_std": 8.659059676574543e-05} {"timestamp_utc": "2026-04-12T03:19:29Z", "mode": "train", "global_step": 3097, "epoch": 0.12439249708800257, "loss": -0.0005, "grad_norm": 0.17411062121391296, "learning_rate": 6.181818181818182e-07, "num_tokens": 7041771.0, "completions/mean_length": 66.375, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.375, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981486797332764, "rewards/meter/std": 1.2580998372868635e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981486797332764, "rewards/total_composite/std": 1.2580998372868635e-05, "reward": 0.9981486797332764, "reward_std": 1.2579364010889549e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00830040592700243, "sampling/sampling_logp_difference/max": 0.5602436065673828, "sampling/importance_sampling_ratio/min": 0.5710699558258057, "sampling/importance_sampling_ratio/mean": 1.0030303001403809, "sampling/importance_sampling_ratio/max": 1.580132246017456, "entropy": 0.05762590793892741, "clip_ratio/low_mean": 0.005681818351149559, "clip_ratio/low_min": 0.005681818351149559, "clip_ratio/high_mean": 0.0018656715983524919, "clip_ratio/high_max": 0.0018656715983524919, "clip_ratio/region_mean": 0.007547489949502051, "reward_total_mean": 0.9981486797332764, "reward_meter_mean": 0.9981486797332764, "reward_meter_std": 1.2580998372868635e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981486797332764, "reward_total_composite_std": 1.2580998372868635e-05} {"timestamp_utc": "2026-04-12T03:19:37Z", "mode": "train", "global_step": 3098, "epoch": 0.12443266256978752, "loss": -0.0254, "grad_norm": 2.0594077110290527, "learning_rate": 6.151515151515152e-07, "num_tokens": 7045692.0, "completions/mean_length": 283.125, "completions/min_length": 262.0, "completions/max_length": 310.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 283.125, "completions/min_terminated_length": 262.0, "completions/max_terminated_length": 310.0, "rewards/meter/mean": 0.9590431451797485, "rewards/meter/std": 0.07114315778017044, "rewards/count_adherence/mean": 0.9722222089767456, "rewards/count_adherence/std": 0.05143444985151291, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.9272365570068359, "rewards/repeat_penalty/std": 0.04068033769726753, "rewards/total_composite/mean": 0.7604771852493286, "rewards/total_composite/std": 0.31678301095962524, "reward": 0.7604771852493286, "reward_std": 0.31678304076194763, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05060786381363869, "sampling/sampling_logp_difference/max": 1.4762558937072754, "sampling/importance_sampling_ratio/min": 0.22849158942699432, "sampling/importance_sampling_ratio/mean": 1.0093165636062622, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.522564172744751, "clip_ratio/low_mean": 0.006949802860617638, "clip_ratio/low_min": 0.006949802860617638, "clip_ratio/high_mean": 0.025927380891516805, "clip_ratio/high_max": 0.025927380891516805, "clip_ratio/region_mean": 0.03287718375213444, "reward_total_mean": 0.7604771852493286, "reward_meter_mean": 0.9590431451797485, "reward_meter_std": 0.07114315778017044, "reward_count_adherence_mean": 0.9722222089767456, "reward_count_adherence_std": 0.05143444985151291, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.9272365570068359, "reward_repeat_penalty_std": 0.04068033769726753, "reward_total_composite_mean": 0.7604771852493286, "reward_total_composite_std": 0.31678301095962524} {"timestamp_utc": "2026-04-12T03:19:43Z", "mode": "train", "global_step": 3099, "epoch": 0.12447282805157248, "loss": 0.0008, "grad_norm": 0.2977856993675232, "learning_rate": 6.121212121212121e-07, "num_tokens": 7047543.0, "completions/mean_length": 71.375, "completions/min_length": 71.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.375, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9994341135025024, "rewards/meter/std": 2.3516733563155867e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994341135025024, "rewards/total_composite/std": 2.3516733563155867e-05, "reward": 0.9994341135025024, "reward_std": 2.3526572476839647e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010309785604476929, "sampling/sampling_logp_difference/max": 1.0300507545471191, "sampling/importance_sampling_ratio/min": 0.3569888472557068, "sampling/importance_sampling_ratio/mean": 0.999859631061554, "sampling/importance_sampling_ratio/max": 1.2391035556793213, "entropy": 0.07044319622218609, "clip_ratio/low_mean": 0.005138941807672381, "clip_ratio/low_min": 0.005138941807672381, "clip_ratio/high_mean": 0.0035211266949772835, "clip_ratio/high_max": 0.0035211266949772835, "clip_ratio/region_mean": 0.008660068502649665, "reward_total_mean": 0.9994341135025024, "reward_meter_mean": 0.9994341135025024, "reward_meter_std": 2.3516733563155867e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994341135025024, "reward_total_composite_std": 2.3516733563155867e-05} {"timestamp_utc": "2026-04-12T03:19:48Z", "mode": "train", "global_step": 3100, "epoch": 0.12451299353335743, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.090909090909092e-07, "num_tokens": 7049375.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 8.230307139456272e-05, "sampling/sampling_logp_difference/max": 0.0053017400205135345, "sampling/importance_sampling_ratio/min": 0.9947122931480408, "sampling/importance_sampling_ratio/mean": 1.0000481605529785, "sampling/importance_sampling_ratio/max": 1.0023247003555298, "entropy": 0.0007583037222502753, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:21:06Z", "mode": "eval", "global_step": 3100, "epoch": 0.12451299353335743, "eval_loss": NaN, "eval_runtime": 77.6639, "eval_samples_per_second": 1.339, "eval_steps_per_second": 0.167, "eval_num_tokens": 7049375.0, "eval_completions/mean_length": 210.7403846153846, "eval_completions/min_length": 61.07692307692308, "eval_completions/max_length": 415.0, "eval_completions/clipped_ratio": 0.019230769230769232, "eval_completions/mean_terminated_length": 204.17994689941406, "eval_completions/min_terminated_length": 61.07692307692308, "eval_completions/max_terminated_length": 392.53846153846155, "eval_rewards/meter/mean": 0.8045936226844788, "eval_rewards/meter/std": 0.3283666269137309, "eval_rewards/count_adherence/mean": 0.9585646253365737, "eval_rewards/count_adherence/std": 0.065789727637401, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/repeat_penalty/mean": 0.9568017308528607, "eval_rewards/repeat_penalty/std": 0.065969582217244, "eval_rewards/total_composite/mean": 0.7422462472548852, "eval_rewards/total_composite/std": 0.3302810134795996, "eval_reward": 0.7422462472548852, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.03276929044379638, "eval_sampling/sampling_logp_difference/max": 1.120217965199397, "eval_sampling/importance_sampling_ratio/min": 0.3349529756949498, "eval_sampling/importance_sampling_ratio/mean": 1.009482246178847, "eval_sampling/importance_sampling_ratio/max": 1.5006029880963838, "eval_entropy": 0.37355728218188655, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7422462472548852, "eval_reward_meter_mean": 0.8045936226844788, "eval_reward_meter_std": 0.3283666269137309, "eval_reward_count_adherence_mean": 0.9585646253365737, "eval_reward_count_adherence_std": 0.065789727637401, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_repeat_penalty_mean": 0.9568017308528607, "eval_reward_repeat_penalty_std": 0.065969582217244, "eval_reward_total_composite_mean": 0.7422462472548852, "eval_reward_total_composite_std": 0.3302810134795996} {"timestamp_utc": "2026-04-12T03:21:16Z", "mode": "train", "global_step": 3101, "epoch": 0.12455315901514238, "loss": 0.0255, "grad_norm": 2.886378765106201, "learning_rate": 6.060606060606061e-07, "num_tokens": 7052434.0, "completions/mean_length": 209.375, "completions/min_length": 198.0, "completions/max_length": 223.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 209.375, "completions/min_terminated_length": 198.0, "completions/max_terminated_length": 223.0, "rewards/meter/mean": 0.9796897768974304, "rewards/meter/std": 0.013765236362814903, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9545454978942871, "rewards/repeat_penalty/std": 0.0971859022974968, "rewards/total_composite/mean": 0.9347763061523438, "rewards/total_composite/std": 0.0924786925315857, "reward": 0.9347763061523438, "reward_std": 0.0924786925315857, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03803591802716255, "sampling/sampling_logp_difference/max": 0.9048151969909668, "sampling/importance_sampling_ratio/min": 0.40461668372154236, "sampling/importance_sampling_ratio/mean": 1.0072675943374634, "sampling/importance_sampling_ratio/max": 1.8678045272827148, "entropy": 0.42885980382561684, "clip_ratio/low_mean": 0.005714374827221036, "clip_ratio/low_min": 0.005714374827221036, "clip_ratio/high_mean": 0.03715289803221822, "clip_ratio/high_max": 0.03715289803221822, "clip_ratio/region_mean": 0.042867272859439254, "reward_total_mean": 0.9347763061523438, "reward_meter_mean": 0.9796897768974304, "reward_meter_std": 0.013765236362814903, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9545454978942871, "reward_repeat_penalty_std": 0.0971859022974968, "reward_total_composite_mean": 0.9347763061523438, "reward_total_composite_std": 0.0924786925315857} {"timestamp_utc": "2026-04-12T03:21:22Z", "mode": "train", "global_step": 3102, "epoch": 0.12459332449692734, "loss": 0.0064, "grad_norm": 3.9274234771728516, "learning_rate": 6.03030303030303e-07, "num_tokens": 7054118.0, "completions/mean_length": 65.5, "completions/min_length": 64.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.5, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9924110770225525, "rewards/meter/std": 0.00258190231397748, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9924110770225525, "rewards/total_composite/std": 0.00258190231397748, "reward": 0.9924110770225525, "reward_std": 0.0025818964932113886, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023297905921936035, "sampling/sampling_logp_difference/max": 1.1882240772247314, "sampling/importance_sampling_ratio/min": 0.30476200580596924, "sampling/importance_sampling_ratio/mean": 1.002340316772461, "sampling/importance_sampling_ratio/max": 1.3347103595733643, "entropy": 0.14634791854768991, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.005800189450383186, "clip_ratio/high_max": 0.005800189450383186, "clip_ratio/region_mean": 0.005800189450383186, "reward_total_mean": 0.9924110770225525, "reward_meter_mean": 0.9924110770225525, "reward_meter_std": 0.00258190231397748, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9924110770225525, "reward_total_composite_std": 0.00258190231397748} {"timestamp_utc": "2026-04-12T03:21:28Z", "mode": "train", "global_step": 3103, "epoch": 0.12463348997871229, "loss": 0.0107, "grad_norm": 5.046589374542236, "learning_rate": 6.000000000000001e-07, "num_tokens": 7056293.0, "completions/mean_length": 85.875, "completions/min_length": 83.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.875, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.9477604627609253, "rewards/meter/std": 0.021915482357144356, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.930683970451355, "rewards/total_composite/std": 0.04980318620800972, "reward": 0.930683970451355, "reward_std": 0.04980317875742912, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06328704953193665, "sampling/sampling_logp_difference/max": 1.8605287075042725, "sampling/importance_sampling_ratio/min": 0.15559034049510956, "sampling/importance_sampling_ratio/mean": 0.993628203868866, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.26804009079933167, "clip_ratio/low_mean": 0.017914682626724243, "clip_ratio/low_min": 0.017914682626724243, "clip_ratio/high_mean": 0.03492570295929909, "clip_ratio/high_max": 0.03492570295929909, "clip_ratio/region_mean": 0.05284038558602333, "reward_total_mean": 0.930683970451355, "reward_meter_mean": 0.9477604627609253, "reward_meter_std": 0.021915482357144356, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.930683970451355, "reward_total_composite_std": 0.04980318620800972} {"timestamp_utc": "2026-04-12T03:21:34Z", "mode": "train", "global_step": 3104, "epoch": 0.12467365546049725, "loss": 0.002, "grad_norm": 3.8720288276672363, "learning_rate": 5.96969696969697e-07, "num_tokens": 7058098.0, "completions/mean_length": 77.625, "completions/min_length": 77.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.625, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9990447163581848, "rewards/meter/std": 0.0004974787007085979, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990447163581848, "rewards/total_composite/std": 0.0004974787007085979, "reward": 0.9990447163581848, "reward_std": 0.0004974827752448618, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019970104098320007, "sampling/sampling_logp_difference/max": 0.8766874074935913, "sampling/importance_sampling_ratio/min": 0.4161592125892639, "sampling/importance_sampling_ratio/mean": 1.0045377016067505, "sampling/importance_sampling_ratio/max": 1.4500758647918701, "entropy": 0.18993182107806206, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.00967782223597169, "clip_ratio/high_max": 0.00967782223597169, "clip_ratio/region_mean": 0.00967782223597169, "reward_total_mean": 0.9990447163581848, "reward_meter_mean": 0.9990447163581848, "reward_meter_std": 0.0004974787007085979, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990447163581848, "reward_total_composite_std": 0.0004974787007085979} {"timestamp_utc": "2026-04-12T03:21:40Z", "mode": "train", "global_step": 3105, "epoch": 0.1247138209422822, "loss": -0.0052, "grad_norm": 6.449222087860107, "learning_rate": 5.93939393939394e-07, "num_tokens": 7059998.0, "completions/mean_length": 66.5, "completions/min_length": 64.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.5, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9450548887252808, "rewards/meter/std": 0.015829402953386307, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9450548887252808, "rewards/total_composite/std": 0.015829402953386307, "reward": 0.9450548887252808, "reward_std": 0.015829408541321754, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03087073564529419, "sampling/sampling_logp_difference/max": 1.6123805046081543, "sampling/importance_sampling_ratio/min": 0.19941234588623047, "sampling/importance_sampling_ratio/mean": 1.0009596347808838, "sampling/importance_sampling_ratio/max": 1.7957602739334106, "entropy": 0.13373739924281836, "clip_ratio/low_mean": 0.007753314450383186, "clip_ratio/low_min": 0.007753314450383186, "clip_ratio/high_mean": 0.02611940260976553, "clip_ratio/high_max": 0.02611940260976553, "clip_ratio/region_mean": 0.033872717060148716, "reward_total_mean": 0.9450548887252808, "reward_meter_mean": 0.9450548887252808, "reward_meter_std": 0.015829402953386307, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9450548887252808, "reward_total_composite_std": 0.015829402953386307} {"timestamp_utc": "2026-04-12T03:21:46Z", "mode": "train", "global_step": 3106, "epoch": 0.12475398642406715, "loss": 0.0012, "grad_norm": 1.8398211002349854, "learning_rate": 5.90909090909091e-07, "num_tokens": 7062542.0, "completions/mean_length": 155.0, "completions/min_length": 153.0, "completions/max_length": 157.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 155.0, "completions/min_terminated_length": 153.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.9989175200462341, "rewards/meter/std": 0.00033335990156047046, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989175200462341, "rewards/total_composite/std": 0.00033335990156047046, "reward": 0.9989175200462341, "reward_std": 0.0003333572531118989, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03886400908231735, "sampling/sampling_logp_difference/max": 2.4166221618652344, "sampling/importance_sampling_ratio/min": 0.0892224907875061, "sampling/importance_sampling_ratio/mean": 1.0071746110916138, "sampling/importance_sampling_ratio/max": 1.7075154781341553, "entropy": 0.3304919973015785, "clip_ratio/low_mean": 0.004859996202867478, "clip_ratio/low_min": 0.004859996202867478, "clip_ratio/high_mean": 0.0121077821822837, "clip_ratio/high_max": 0.0121077821822837, "clip_ratio/region_mean": 0.016967778385151178, "reward_total_mean": 0.9989175200462341, "reward_meter_mean": 0.9989175200462341, "reward_meter_std": 0.00033335990156047046, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9989175200462341, "reward_total_composite_std": 0.00033335990156047046} {"timestamp_utc": "2026-04-12T03:21:56Z", "mode": "train", "global_step": 3107, "epoch": 0.12479415190585211, "loss": -0.0223, "grad_norm": 1.4729881286621094, "learning_rate": 5.878787878787879e-07, "num_tokens": 7067381.0, "completions/mean_length": 403.875, "completions/min_length": 383.0, "completions/max_length": 426.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 403.875, "completions/min_terminated_length": 383.0, "completions/max_terminated_length": 426.0, "rewards/meter/mean": 0.9989637136459351, "rewards/meter/std": 0.00037519802572205663, "rewards/count_adherence/mean": 0.7980769276618958, "rewards/count_adherence/std": 0.039811473339796066, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9746710062026978, "rewards/repeat_penalty/std": 0.038055676966905594, "rewards/total_composite/mean": 0.7777890563011169, "rewards/total_composite/std": 0.061021268367767334, "reward": 0.7777890563011169, "reward_std": 0.061021264642477036, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04732804745435715, "sampling/sampling_logp_difference/max": 1.5902900695800781, "sampling/importance_sampling_ratio/min": 0.20386646687984467, "sampling/importance_sampling_ratio/mean": 1.0116568803787231, "sampling/importance_sampling_ratio/max": 1.9504014253616333, "entropy": 0.4669685959815979, "clip_ratio/low_mean": 0.01397907012142241, "clip_ratio/low_min": 0.01397907012142241, "clip_ratio/high_mean": 0.014827882871031761, "clip_ratio/high_max": 0.014827882871031761, "clip_ratio/region_mean": 0.02880695299245417, "reward_total_mean": 0.7777890563011169, "reward_meter_mean": 0.9989637136459351, "reward_meter_std": 0.00037519802572205663, "reward_count_adherence_mean": 0.7980769276618958, "reward_count_adherence_std": 0.039811473339796066, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9746710062026978, "reward_repeat_penalty_std": 0.038055676966905594, "reward_total_composite_mean": 0.7777890563011169, "reward_total_composite_std": 0.061021268367767334} {"timestamp_utc": "2026-04-12T03:22:01Z", "mode": "train", "global_step": 3108, "epoch": 0.12483431738763706, "loss": 0.0018, "grad_norm": 0.7012980580329895, "learning_rate": 5.848484848484849e-07, "num_tokens": 7069454.0, "completions/mean_length": 93.125, "completions/min_length": 93.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.125, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.9972705841064453, "rewards/meter/std": 0.0013248658506199718, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9972705841064453, "rewards/total_composite/std": 0.0013248658506199718, "reward": 0.9972705841064453, "reward_std": 0.0013248566538095474, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009307874366641045, "sampling/sampling_logp_difference/max": 0.49368953704833984, "sampling/importance_sampling_ratio/min": 0.6166864037513733, "sampling/importance_sampling_ratio/mean": 1.004794716835022, "sampling/importance_sampling_ratio/max": 1.6383498907089233, "entropy": 0.0752844586968422, "clip_ratio/low_mean": 0.003989361692219973, "clip_ratio/low_min": 0.003989361692219973, "clip_ratio/high_mean": 0.0067204301012679935, "clip_ratio/high_max": 0.0067204301012679935, "clip_ratio/region_mean": 0.010709791793487966, "reward_total_mean": 0.9972705841064453, "reward_meter_mean": 0.9972705841064453, "reward_meter_std": 0.0013248658506199718, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9972705841064453, "reward_total_composite_std": 0.0013248658506199718} {"timestamp_utc": "2026-04-12T03:22:08Z", "mode": "train", "global_step": 3109, "epoch": 0.12487448286942202, "loss": 0.0016, "grad_norm": 3.0335049629211426, "learning_rate": 5.818181818181819e-07, "num_tokens": 7072207.0, "completions/mean_length": 172.125, "completions/min_length": 165.0, "completions/max_length": 178.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 172.125, "completions/min_terminated_length": 165.0, "completions/max_terminated_length": 178.0, "rewards/meter/mean": 0.961860179901123, "rewards/meter/std": 0.08823229372501373, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.9342338442802429, "rewards/total_composite/std": 0.0913098156452179, "reward": 0.9342338442802429, "reward_std": 0.09130982309579849, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03732313588261604, "sampling/sampling_logp_difference/max": 2.1083364486694336, "sampling/importance_sampling_ratio/min": 0.12143982201814651, "sampling/importance_sampling_ratio/mean": 1.0083178281784058, "sampling/importance_sampling_ratio/max": 1.8755393028259277, "entropy": 0.37529650889337063, "clip_ratio/low_mean": 0.011014241958037019, "clip_ratio/low_min": 0.011014241958037019, "clip_ratio/high_mean": 0.029623867012560368, "clip_ratio/high_max": 0.029623867012560368, "clip_ratio/region_mean": 0.040638108970597386, "reward_total_mean": 0.9342338442802429, "reward_meter_mean": 0.961860179901123, "reward_meter_std": 0.08823229372501373, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_total_composite_mean": 0.9342338442802429, "reward_total_composite_std": 0.0913098156452179} {"timestamp_utc": "2026-04-12T03:22:14Z", "mode": "train", "global_step": 3110, "epoch": 0.12491464835120697, "loss": 0.0026, "grad_norm": 2.6455962657928467, "learning_rate": 5.787878787878789e-07, "num_tokens": 7074959.0, "completions/mean_length": 166.0, "completions/min_length": 162.0, "completions/max_length": 170.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 166.0, "completions/min_terminated_length": 162.0, "completions/max_terminated_length": 170.0, "rewards/meter/mean": 0.9984124898910522, "rewards/meter/std": 0.0009795344667509198, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9984124898910522, "rewards/total_composite/std": 0.0009795344667509198, "reward": 0.9984124898910522, "reward_std": 0.0009795371443033218, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04640757292509079, "sampling/sampling_logp_difference/max": 3.5209038257598877, "sampling/importance_sampling_ratio/min": 0.02957269363105297, "sampling/importance_sampling_ratio/mean": 1.0051828622817993, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3597927987575531, "clip_ratio/low_mean": 0.004545454401522875, "clip_ratio/low_min": 0.004545454401522875, "clip_ratio/high_mean": 0.03839668887667358, "clip_ratio/high_max": 0.03839668887667358, "clip_ratio/region_mean": 0.042942143278196454, "reward_total_mean": 0.9984124898910522, "reward_meter_mean": 0.9984124898910522, "reward_meter_std": 0.0009795344667509198, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9984124898910522, "reward_total_composite_std": 0.0009795344667509198} {"timestamp_utc": "2026-04-12T03:22:18Z", "mode": "train", "global_step": 3111, "epoch": 0.12495481383299192, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.757575757575758e-07, "num_tokens": 7076575.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00021090851805638522, "sampling/sampling_logp_difference/max": 0.0025518755428493023, "sampling/importance_sampling_ratio/min": 0.9997379779815674, "sampling/importance_sampling_ratio/mean": 1.0002080202102661, "sampling/importance_sampling_ratio/max": 1.0025551319122314, "entropy": 0.0017129705229308456, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:22:26Z", "mode": "train", "global_step": 3112, "epoch": 0.12499497931477688, "loss": 0.0147, "grad_norm": 2.5759596824645996, "learning_rate": 5.727272727272728e-07, "num_tokens": 7080735.0, "completions/mean_length": 303.0, "completions/min_length": 286.0, "completions/max_length": 313.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 303.0, "completions/min_terminated_length": 286.0, "completions/max_terminated_length": 313.0, "rewards/meter/mean": 0.9937278628349304, "rewards/meter/std": 0.004818979650735855, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9937278628349304, "rewards/total_composite/std": 0.004818979650735855, "reward": 0.9937278628349304, "reward_std": 0.004818981513381004, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0645064041018486, "sampling/sampling_logp_difference/max": 6.46595573425293, "sampling/importance_sampling_ratio/min": 0.0015555039281025529, "sampling/importance_sampling_ratio/mean": 1.0059781074523926, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5345265790820122, "clip_ratio/low_mean": 0.022525052540004253, "clip_ratio/low_min": 0.022525052540004253, "clip_ratio/high_mean": 0.026788415852934122, "clip_ratio/high_max": 0.026788415852934122, "clip_ratio/region_mean": 0.049313468392938375, "reward_total_mean": 0.9937278628349304, "reward_meter_mean": 0.9937278628349304, "reward_meter_std": 0.004818979650735855, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9937278628349304, "reward_total_composite_std": 0.004818979650735855} {"timestamp_utc": "2026-04-12T03:22:30Z", "mode": "train", "global_step": 3113, "epoch": 0.12503514479656183, "loss": -0.0003, "grad_norm": 0.42850980162620544, "learning_rate": 5.696969696969698e-07, "num_tokens": 7082156.0, "completions/mean_length": 40.625, "completions/min_length": 39.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.625, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.998562216758728, "rewards/meter/std": 0.0005881982506252825, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998562216758728, "rewards/total_composite/std": 0.0005881982506252825, "reward": 0.998562216758728, "reward_std": 0.0005882052937522531, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009273692034184933, "sampling/sampling_logp_difference/max": 0.7357292175292969, "sampling/importance_sampling_ratio/min": 0.47915592789649963, "sampling/importance_sampling_ratio/mean": 1.0016939640045166, "sampling/importance_sampling_ratio/max": 1.4733530282974243, "entropy": 0.09460031241178513, "clip_ratio/low_mean": 0.0062500000931322575, "clip_ratio/low_min": 0.0062500000931322575, "clip_ratio/high_mean": 0.006097560748457909, "clip_ratio/high_max": 0.006097560748457909, "clip_ratio/region_mean": 0.012347560841590166, "reward_total_mean": 0.998562216758728, "reward_meter_mean": 0.998562216758728, "reward_meter_std": 0.0005881982506252825, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.998562216758728, "reward_total_composite_std": 0.0005881982506252825} {"timestamp_utc": "2026-04-12T03:22:35Z", "mode": "train", "global_step": 3114, "epoch": 0.12507531027834679, "loss": -0.0061, "grad_norm": 1.817691683769226, "learning_rate": 5.666666666666667e-07, "num_tokens": 7084326.0, "completions/mean_length": 97.25, "completions/min_length": 95.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.25, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9978147745132446, "rewards/meter/std": 0.00044735919800587, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978147745132446, "rewards/total_composite/std": 0.00044735919800587, "reward": 0.9978147745132446, "reward_std": 0.0004473614681046456, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013976364396512508, "sampling/sampling_logp_difference/max": 0.9773850440979004, "sampling/importance_sampling_ratio/min": 0.3762938380241394, "sampling/importance_sampling_ratio/mean": 1.0041371583938599, "sampling/importance_sampling_ratio/max": 1.3954778909683228, "entropy": 0.10958225093781948, "clip_ratio/low_mean": 0.0013157895300537348, "clip_ratio/low_min": 0.0013157895300537348, "clip_ratio/high_mean": 0.008981169550679624, "clip_ratio/high_max": 0.008981169550679624, "clip_ratio/region_mean": 0.010296959080733359, "reward_total_mean": 0.9978147745132446, "reward_meter_mean": 0.9978147745132446, "reward_meter_std": 0.00044735919800587, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9978147745132446, "reward_total_composite_std": 0.00044735919800587} {"timestamp_utc": "2026-04-12T03:22:39Z", "mode": "train", "global_step": 3115, "epoch": 0.12511547576013174, "loss": 0.0004, "grad_norm": 0.015785139054059982, "learning_rate": 5.636363636363638e-07, "num_tokens": 7086062.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9973392486572266, "rewards/meter/std": 4.345127990745823e-07, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973392486572266, "rewards/total_composite/std": 4.345127990745823e-07, "reward": 0.9973392486572266, "reward_std": 4.280403231859964e-07, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004269226919859648, "sampling/sampling_logp_difference/max": 0.5515934228897095, "sampling/importance_sampling_ratio/min": 0.5760312080383301, "sampling/importance_sampling_ratio/mean": 0.9995636940002441, "sampling/importance_sampling_ratio/max": 1.2580742835998535, "entropy": 0.02565961377695203, "clip_ratio/low_mean": 0.0020491802133619785, "clip_ratio/low_min": 0.0020491802133619785, "clip_ratio/high_mean": 0.0020491802133619785, "clip_ratio/high_max": 0.0020491802133619785, "clip_ratio/region_mean": 0.004098360426723957, "reward_total_mean": 0.9973392486572266, "reward_meter_mean": 0.9973392486572266, "reward_meter_std": 4.345127990745823e-07, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9973392486572266, "reward_total_composite_std": 4.345127990745823e-07} {"timestamp_utc": "2026-04-12T03:22:50Z", "mode": "train", "global_step": 3116, "epoch": 0.1251556412419167, "loss": 0.0232, "grad_norm": 1.0452237129211426, "learning_rate": 5.606060606060607e-07, "num_tokens": 7091420.0, "completions/mean_length": 490.75, "completions/min_length": 461.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 487.71429443359375, "completions/min_terminated_length": 461.0, "completions/max_terminated_length": 509.0, "rewards/meter/mean": 0.9985963106155396, "rewards/meter/std": 0.0007045524544082582, "rewards/count_adherence/mean": 0.8515625, "rewards/count_adherence/std": 0.032346826046705246, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9714829921722412, "rewards/repeat_penalty/std": 0.027092305943369865, "rewards/total_composite/mean": 0.8266471028327942, "rewards/total_composite/std": 0.049881432205438614, "reward": 0.8266471028327942, "reward_std": 0.049881432205438614, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.034601159393787384, "sampling/sampling_logp_difference/max": 1.3683700561523438, "sampling/importance_sampling_ratio/min": 0.2545214593410492, "sampling/importance_sampling_ratio/mean": 1.0090035200119019, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3404971808195114, "clip_ratio/low_mean": 0.0044667941983789206, "clip_ratio/low_min": 0.0044667941983789206, "clip_ratio/high_mean": 0.013212713180109859, "clip_ratio/high_max": 0.013212713180109859, "clip_ratio/region_mean": 0.01767950737848878, "reward_total_mean": 0.8266471028327942, "reward_meter_mean": 0.9985963106155396, "reward_meter_std": 0.0007045524544082582, "reward_count_adherence_mean": 0.8515625, "reward_count_adherence_std": 0.032346826046705246, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9714829921722412, "reward_repeat_penalty_std": 0.027092305943369865, "reward_total_composite_mean": 0.8266471028327942, "reward_total_composite_std": 0.049881432205438614} {"timestamp_utc": "2026-04-12T03:22:54Z", "mode": "train", "global_step": 3117, "epoch": 0.12519580672370165, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.575757575757576e-07, "num_tokens": 7092924.0, "completions/mean_length": 34.0, "completions/min_length": 34.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9923644065856934, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9923644065856934, "rewards/total_composite/std": 0.0, "reward": 0.9923644065856934, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.004286302253603935, "sampling/sampling_logp_difference/max": 0.12234780192375183, "sampling/importance_sampling_ratio/min": 0.9824780821800232, "sampling/importance_sampling_ratio/mean": 1.0040427446365356, "sampling/importance_sampling_ratio/max": 1.1301469802856445, "entropy": 0.03289771918207407, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9923644065856934, "reward_meter_mean": 0.9923644065856934, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9923644065856934, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:23:02Z", "mode": "train", "global_step": 3118, "epoch": 0.1252359722054866, "loss": 0.0017, "grad_norm": 1.1702804565429688, "learning_rate": 5.545454545454547e-07, "num_tokens": 7097281.0, "completions/mean_length": 354.625, "completions/min_length": 350.0, "completions/max_length": 359.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 354.625, "completions/min_terminated_length": 350.0, "completions/max_terminated_length": 359.0, "rewards/meter/mean": 0.9987514019012451, "rewards/meter/std": 0.00020078917441423982, "rewards/count_adherence/mean": 0.9090909361839294, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9671052694320679, "rewards/repeat_penalty/std": 0.027239417657256126, "rewards/total_composite/mean": 0.8780869841575623, "rewards/total_composite/std": 0.024653173983097076, "reward": 0.8780869841575623, "reward_std": 0.02465316839516163, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.029794396832585335, "sampling/sampling_logp_difference/max": 1.3524599075317383, "sampling/importance_sampling_ratio/min": 0.2586033344268799, "sampling/importance_sampling_ratio/mean": 1.0095971822738647, "sampling/importance_sampling_ratio/max": 1.7627272605895996, "entropy": 0.3398951441049576, "clip_ratio/low_mean": 0.008492077584378421, "clip_ratio/low_min": 0.008492077584378421, "clip_ratio/high_mean": 0.008788066916167736, "clip_ratio/high_max": 0.008788066916167736, "clip_ratio/region_mean": 0.017280144500546157, "reward_total_mean": 0.8780869841575623, "reward_meter_mean": 0.9987514019012451, "reward_meter_std": 0.00020078917441423982, "reward_count_adherence_mean": 0.9090909361839294, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9671052694320679, "reward_repeat_penalty_std": 0.027239417657256126, "reward_total_composite_mean": 0.8780869841575623, "reward_total_composite_std": 0.024653173983097076} {"timestamp_utc": "2026-04-12T03:23:07Z", "mode": "train", "global_step": 3119, "epoch": 0.12527613768727155, "loss": -0.0005, "grad_norm": 5.017796516418457, "learning_rate": 5.515151515151516e-07, "num_tokens": 7098968.0, "completions/mean_length": 60.875, "completions/min_length": 60.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.875, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9972956776618958, "rewards/meter/std": 0.0001135577549575828, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9972956776618958, "rewards/total_composite/std": 0.0001135577549575828, "reward": 0.9972956776618958, "reward_std": 0.0001135657585109584, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005037278868257999, "sampling/sampling_logp_difference/max": 1.2046854496002197, "sampling/importance_sampling_ratio/min": 0.2997862994670868, "sampling/importance_sampling_ratio/mean": 0.9997225999832153, "sampling/importance_sampling_ratio/max": 1.1826646327972412, "entropy": 0.02145107788965106, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.004098360426723957, "clip_ratio/high_max": 0.004098360426723957, "clip_ratio/region_mean": 0.004098360426723957, "reward_total_mean": 0.9972956776618958, "reward_meter_mean": 0.9972956776618958, "reward_meter_std": 0.0001135577549575828, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9972956776618958, "reward_total_composite_std": 0.0001135577549575828} {"timestamp_utc": "2026-04-12T03:23:12Z", "mode": "train", "global_step": 3120, "epoch": 0.1253163031690565, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.484848484848485e-07, "num_tokens": 7100936.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00018837934476323426, "sampling/sampling_logp_difference/max": 0.0016895392909646034, "sampling/importance_sampling_ratio/min": 0.998882532119751, "sampling/importance_sampling_ratio/mean": 1.0001810789108276, "sampling/importance_sampling_ratio/max": 1.0016909837722778, "entropy": 0.0014699531602673233, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:23:21Z", "mode": "train", "global_step": 3121, "epoch": 0.12535646865084146, "loss": 0.0239, "grad_norm": 1.84979248046875, "learning_rate": 5.454545454545455e-07, "num_tokens": 7105325.0, "completions/mean_length": 354.625, "completions/min_length": 342.0, "completions/max_length": 394.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 354.625, "completions/min_terminated_length": 342.0, "completions/max_terminated_length": 394.0, "rewards/meter/mean": 0.9991722106933594, "rewards/meter/std": 0.0001384944043820724, "rewards/count_adherence/mean": 0.9124999642372131, "rewards/count_adherence/std": 0.0353553481400013, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6744744777679443, "rewards/total_composite/std": 0.41629472374916077, "reward": 0.6744744777679443, "reward_std": 0.4162946939468384, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04996606335043907, "sampling/sampling_logp_difference/max": 2.3657424449920654, "sampling/importance_sampling_ratio/min": 0.0938795730471611, "sampling/importance_sampling_ratio/mean": 1.0077815055847168, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47248122096061707, "clip_ratio/low_mean": 0.007510315626859665, "clip_ratio/low_min": 0.007510315626859665, "clip_ratio/high_mean": 0.02431198745034635, "clip_ratio/high_max": 0.02431198745034635, "clip_ratio/region_mean": 0.031822303077206016, "reward_total_mean": 0.6744744777679443, "reward_meter_mean": 0.9991722106933594, "reward_meter_std": 0.0001384944043820724, "reward_count_adherence_mean": 0.9124999642372131, "reward_count_adherence_std": 0.0353553481400013, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6744744777679443, "reward_total_composite_std": 0.41629472374916077} {"timestamp_utc": "2026-04-12T03:23:26Z", "mode": "train", "global_step": 3122, "epoch": 0.12539663413262642, "loss": 0.0001, "grad_norm": 0.5849205851554871, "learning_rate": 5.424242424242425e-07, "num_tokens": 7107085.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9973349571228027, "rewards/meter/std": 1.3741479051532224e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973349571228027, "rewards/total_composite/std": 1.3741479051532224e-05, "reward": 0.9973349571228027, "reward_std": 1.3751973710895982e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005648355465382338, "sampling/sampling_logp_difference/max": 0.814976692199707, "sampling/importance_sampling_ratio/min": 0.44264963269233704, "sampling/importance_sampling_ratio/mean": 0.9968710541725159, "sampling/importance_sampling_ratio/max": 1.0885894298553467, "entropy": 0.019400187535211444, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.006147540640085936, "clip_ratio/high_max": 0.006147540640085936, "clip_ratio/region_mean": 0.006147540640085936, "reward_total_mean": 0.9973349571228027, "reward_meter_mean": 0.9973349571228027, "reward_meter_std": 1.3741479051532224e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9973349571228027, "reward_total_composite_std": 1.3741479051532224e-05} {"timestamp_utc": "2026-04-12T03:23:32Z", "mode": "train", "global_step": 3123, "epoch": 0.12543679961441137, "loss": 0.0044, "grad_norm": 2.6911187171936035, "learning_rate": 5.393939393939395e-07, "num_tokens": 7109065.0, "completions/mean_length": 100.5, "completions/min_length": 99.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.5, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.9990350008010864, "rewards/meter/std": 0.0004222920979373157, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990350008010864, "rewards/total_composite/std": 0.0004222920979373157, "reward": 0.9990350008010864, "reward_std": 0.0004223074938636273, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025933468714356422, "sampling/sampling_logp_difference/max": 0.8458013534545898, "sampling/importance_sampling_ratio/min": 0.5429390072822571, "sampling/importance_sampling_ratio/mean": 1.0030157566070557, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1784425787627697, "clip_ratio/low_mean": 0.007426470518112183, "clip_ratio/low_min": 0.007426470518112183, "clip_ratio/high_mean": 0.02493911876808852, "clip_ratio/high_max": 0.02493911876808852, "clip_ratio/region_mean": 0.0323655892862007, "reward_total_mean": 0.9990350008010864, "reward_meter_mean": 0.9990350008010864, "reward_meter_std": 0.0004222920979373157, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990350008010864, "reward_total_composite_std": 0.0004222920979373157} {"timestamp_utc": "2026-04-12T03:23:42Z", "mode": "train", "global_step": 3124, "epoch": 0.12547696509619632, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.363636363636364e-07, "num_tokens": 7110721.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9872919917106628, "rewards/meter/std": 0.03354829549789429, "rewards/count_adherence/mean": 0.7361111044883728, "rewards/count_adherence/std": 0.0257172379642725, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9615384340286255, "rewards/repeat_penalty/std": 0.029074188321828842, "rewards/total_composite/mean": 0.6985058784484863, "rewards/total_composite/std": 0.03456052392721176, "reward": 0.6985058784484863, "reward_std": 0.03456050530076027, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.6985058784484863, "reward_meter_mean": 0.9872919917106628, "reward_meter_std": 0.03354829549789429, "reward_count_adherence_mean": 0.7361111044883728, "reward_count_adherence_std": 0.0257172379642725, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9615384340286255, "reward_repeat_penalty_std": 0.029074188321828842, "reward_total_composite_mean": 0.6985058784484863, "reward_total_composite_std": 0.03456052392721176} {"timestamp_utc": "2026-04-12T03:23:47Z", "mode": "train", "global_step": 3125, "epoch": 0.12551713057798128, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.333333333333335e-07, "num_tokens": 7112433.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00018432487559039146, "sampling/sampling_logp_difference/max": 0.0014849056024104357, "sampling/importance_sampling_ratio/min": 0.9985162019729614, "sampling/importance_sampling_ratio/mean": 1.0001718997955322, "sampling/importance_sampling_ratio/max": 1.0014132261276245, "entropy": 0.0014961521519580856, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:23:56Z", "mode": "train", "global_step": 3126, "epoch": 0.12555729605976623, "loss": -0.0003, "grad_norm": 1.4976996183395386, "learning_rate": 5.303030303030304e-07, "num_tokens": 7117396.0, "completions/mean_length": 418.375, "completions/min_length": 411.0, "completions/max_length": 423.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 418.375, "completions/min_terminated_length": 411.0, "completions/max_terminated_length": 423.0, "rewards/meter/mean": 0.9984115362167358, "rewards/meter/std": 0.0010889971163123846, "rewards/count_adherence/mean": 0.8365384340286255, "rewards/count_adherence/std": 0.027196412906050682, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.988095223903656, "rewards/repeat_penalty/std": 0.033671751618385315, "rewards/total_composite/mean": 0.8251492977142334, "rewards/total_composite/std": 0.03634733706712723, "reward": 0.8251492977142334, "reward_std": 0.036347318440675735, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05572379007935524, "sampling/sampling_logp_difference/max": 1.5548510551452637, "sampling/importance_sampling_ratio/min": 0.21122083067893982, "sampling/importance_sampling_ratio/mean": 1.0117367506027222, "sampling/importance_sampling_ratio/max": 1.753551959991455, "entropy": 0.51447868719697, "clip_ratio/low_mean": 0.007205356378108263, "clip_ratio/low_min": 0.007205356378108263, "clip_ratio/high_mean": 0.026507085654884577, "clip_ratio/high_max": 0.026507085654884577, "clip_ratio/region_mean": 0.03371244203299284, "reward_total_mean": 0.8251492977142334, "reward_meter_mean": 0.9984115362167358, "reward_meter_std": 0.0010889971163123846, "reward_count_adherence_mean": 0.8365384340286255, "reward_count_adherence_std": 0.027196412906050682, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.988095223903656, "reward_repeat_penalty_std": 0.033671751618385315, "reward_total_composite_mean": 0.8251492977142334, "reward_total_composite_std": 0.03634733706712723} {"timestamp_utc": "2026-04-12T03:24:06Z", "mode": "train", "global_step": 3127, "epoch": 0.1255974615415512, "loss": -0.0212, "grad_norm": 1.312803030014038, "learning_rate": 5.272727272727273e-07, "num_tokens": 7122682.0, "completions/mean_length": 429.75, "completions/min_length": 418.0, "completions/max_length": 468.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 429.75, "completions/min_terminated_length": 418.0, "completions/max_terminated_length": 468.0, "rewards/meter/mean": 0.9992569088935852, "rewards/meter/std": 7.736113184364513e-05, "rewards/count_adherence/mean": 0.7946428060531616, "rewards/count_adherence/std": 0.025253823027014732, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9824134111404419, "rewards/repeat_penalty/std": 0.024280980229377747, "rewards/total_composite/mean": 0.7802447080612183, "rewards/total_composite/std": 0.035804178565740585, "reward": 0.7802447080612183, "reward_std": 0.03580417484045029, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05896211043000221, "sampling/sampling_logp_difference/max": 10.541126251220703, "sampling/importance_sampling_ratio/min": 2.6426951080793515e-05, "sampling/importance_sampling_ratio/mean": 1.0106730461120605, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.49032482132315636, "clip_ratio/low_mean": 0.00941323209553957, "clip_ratio/low_min": 0.00941323209553957, "clip_ratio/high_mean": 0.022230799309909344, "clip_ratio/high_max": 0.022230799309909344, "clip_ratio/region_mean": 0.031644031405448914, "reward_total_mean": 0.7802447080612183, "reward_meter_mean": 0.9992569088935852, "reward_meter_std": 7.736113184364513e-05, "reward_count_adherence_mean": 0.7946428060531616, "reward_count_adherence_std": 0.025253823027014732, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9824134111404419, "reward_repeat_penalty_std": 0.024280980229377747, "reward_total_composite_mean": 0.7802447080612183, "reward_total_composite_std": 0.035804178565740585} {"timestamp_utc": "2026-04-12T03:24:12Z", "mode": "train", "global_step": 3128, "epoch": 0.12563762702333614, "loss": 0.0112, "grad_norm": 1.9481908082962036, "learning_rate": 5.242424242424243e-07, "num_tokens": 7125754.0, "completions/mean_length": 186.0, "completions/min_length": 182.0, "completions/max_length": 189.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 186.0, "completions/min_terminated_length": 182.0, "completions/max_terminated_length": 189.0, "rewards/meter/mean": 0.9975753426551819, "rewards/meter/std": 0.000392713351175189, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9318182468414307, "rewards/repeat_penalty/std": 0.04208271950483322, "rewards/total_composite/mean": 0.9295545816421509, "rewards/total_composite/std": 0.041869208216667175, "reward": 0.9295545816421509, "reward_std": 0.04186920449137688, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031066881492733955, "sampling/sampling_logp_difference/max": 1.1641087532043457, "sampling/importance_sampling_ratio/min": 0.31220078468322754, "sampling/importance_sampling_ratio/mean": 1.00235915184021, "sampling/importance_sampling_ratio/max": 1.669154405593872, "entropy": 0.2618657611310482, "clip_ratio/low_mean": 0.010725616884883493, "clip_ratio/low_min": 0.010725616884883493, "clip_ratio/high_mean": 0.010851842351257801, "clip_ratio/high_max": 0.010851842351257801, "clip_ratio/region_mean": 0.021577459236141294, "reward_total_mean": 0.9295545816421509, "reward_meter_mean": 0.9975753426551819, "reward_meter_std": 0.000392713351175189, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9318182468414307, "reward_repeat_penalty_std": 0.04208271950483322, "reward_total_composite_mean": 0.9295545816421509, "reward_total_composite_std": 0.041869208216667175} {"timestamp_utc": "2026-04-12T03:24:18Z", "mode": "train", "global_step": 3129, "epoch": 0.1256777925051211, "loss": 0.0067, "grad_norm": 2.561624526977539, "learning_rate": 5.212121212121213e-07, "num_tokens": 7128173.0, "completions/mean_length": 127.375, "completions/min_length": 123.0, "completions/max_length": 130.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.375, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.9965527057647705, "rewards/meter/std": 0.0038294827099889517, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9787326455116272, "rewards/total_composite/std": 0.04999339208006859, "reward": 0.9787326455116272, "reward_std": 0.04999341070652008, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017186593264341354, "sampling/sampling_logp_difference/max": 1.2021827697753906, "sampling/importance_sampling_ratio/min": 0.30053746700286865, "sampling/importance_sampling_ratio/mean": 1.0033414363861084, "sampling/importance_sampling_ratio/max": 1.5255719423294067, "entropy": 0.14594370871782303, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.011735557112842798, "clip_ratio/high_max": 0.011735557112842798, "clip_ratio/region_mean": 0.011735557112842798, "reward_total_mean": 0.9787326455116272, "reward_meter_mean": 0.9965527057647705, "reward_meter_std": 0.0038294827099889517, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.9787326455116272, "reward_total_composite_std": 0.04999339208006859} {"timestamp_utc": "2026-04-12T03:24:22Z", "mode": "train", "global_step": 3130, "epoch": 0.12571795798690605, "loss": 0.0179, "grad_norm": 3.1293835639953613, "learning_rate": 5.181818181818182e-07, "num_tokens": 7129971.0, "completions/mean_length": 65.75, "completions/min_length": 64.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.75, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9895097613334656, "rewards/meter/std": 0.006478929426521063, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9895097613334656, "rewards/total_composite/std": 0.006478929426521063, "reward": 0.9895097613334656, "reward_std": 0.006478935480117798, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02998344786465168, "sampling/sampling_logp_difference/max": 0.9678926467895508, "sampling/importance_sampling_ratio/min": 0.37988272309303284, "sampling/importance_sampling_ratio/mean": 1.008150577545166, "sampling/importance_sampling_ratio/max": 1.572881817817688, "entropy": 0.23753815703094006, "clip_ratio/low_mean": 0.020387701224535704, "clip_ratio/low_min": 0.020387701224535704, "clip_ratio/high_mean": 0.007722355774603784, "clip_ratio/high_max": 0.007722355774603784, "clip_ratio/region_mean": 0.028110056999139488, "reward_total_mean": 0.9895097613334656, "reward_meter_mean": 0.9895097613334656, "reward_meter_std": 0.006478929426521063, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9895097613334656, "reward_total_composite_std": 0.006478929426521063} {"timestamp_utc": "2026-04-12T03:24:27Z", "mode": "train", "global_step": 3131, "epoch": 0.125758123468691, "loss": 0.0024, "grad_norm": 3.762568473815918, "learning_rate": 5.151515151515152e-07, "num_tokens": 7131734.0, "completions/mean_length": 65.375, "completions/min_length": 65.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.375, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9759130477905273, "rewards/meter/std": 0.04954683780670166, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9759130477905273, "rewards/total_composite/std": 0.04954683780670166, "reward": 0.9759130477905273, "reward_std": 0.04954684525728226, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020955001935362816, "sampling/sampling_logp_difference/max": 1.4279863834381104, "sampling/importance_sampling_ratio/min": 0.32048511505126953, "sampling/importance_sampling_ratio/mean": 1.0054841041564941, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10212477203458548, "clip_ratio/low_mean": 0.005769230891019106, "clip_ratio/low_min": 0.005769230891019106, "clip_ratio/high_mean": 0.009557109675370157, "clip_ratio/high_max": 0.009557109675370157, "clip_ratio/region_mean": 0.015326340566389263, "reward_total_mean": 0.9759130477905273, "reward_meter_mean": 0.9759130477905273, "reward_meter_std": 0.04954683780670166, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9759130477905273, "reward_total_composite_std": 0.04954683780670166} {"timestamp_utc": "2026-04-12T03:24:32Z", "mode": "train", "global_step": 3132, "epoch": 0.12579828895047596, "loss": 0.0002, "grad_norm": 0.13032668828964233, "learning_rate": 5.121212121212121e-07, "num_tokens": 7133676.0, "completions/mean_length": 66.75, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.75, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981508255004883, "rewards/meter/std": 6.344345820252784e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981508255004883, "rewards/total_composite/std": 6.344345820252784e-06, "reward": 0.9981508255004883, "reward_std": 6.347765065584099e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007171308156102896, "sampling/sampling_logp_difference/max": 0.5747629404067993, "sampling/importance_sampling_ratio/min": 0.5628383159637451, "sampling/importance_sampling_ratio/mean": 1.0008666515350342, "sampling/importance_sampling_ratio/max": 1.516728162765503, "entropy": 0.05024628387764096, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/high_mean": 0.0037313431967049837, "clip_ratio/high_max": 0.0037313431967049837, "clip_ratio/region_mean": 0.005597014795057476, "reward_total_mean": 0.9981508255004883, "reward_meter_mean": 0.9981508255004883, "reward_meter_std": 6.344345820252784e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981508255004883, "reward_total_composite_std": 6.344345820252784e-06} {"timestamp_utc": "2026-04-12T03:24:37Z", "mode": "train", "global_step": 3133, "epoch": 0.1258384544322609, "loss": 0.0045, "grad_norm": 2.7664103507995605, "learning_rate": 5.090909090909092e-07, "num_tokens": 7135896.0, "completions/mean_length": 100.5, "completions/min_length": 99.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.5, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.999111533164978, "rewards/meter/std": 0.00020586077880579978, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999111533164978, "rewards/total_composite/std": 0.00020586077880579978, "reward": 0.999111533164978, "reward_std": 0.00020586182654369622, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025510529056191444, "sampling/sampling_logp_difference/max": 1.5977736711502075, "sampling/importance_sampling_ratio/min": 0.20234650373458862, "sampling/importance_sampling_ratio/mean": 1.0036976337432861, "sampling/importance_sampling_ratio/max": 1.8379536867141724, "entropy": 0.17687924578785896, "clip_ratio/low_mean": 0.00993861141614616, "clip_ratio/low_min": 0.00993861141614616, "clip_ratio/high_mean": 0.006225490127690136, "clip_ratio/high_max": 0.006225490127690136, "clip_ratio/region_mean": 0.016164101543836296, "reward_total_mean": 0.999111533164978, "reward_meter_mean": 0.999111533164978, "reward_meter_std": 0.00020586077880579978, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.999111533164978, "reward_total_composite_std": 0.00020586077880579978} {"timestamp_utc": "2026-04-12T03:24:42Z", "mode": "train", "global_step": 3134, "epoch": 0.12587861991404586, "loss": 0.0178, "grad_norm": 5.516884803771973, "learning_rate": 5.060606060606061e-07, "num_tokens": 7138289.0, "completions/mean_length": 110.125, "completions/min_length": 107.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 110.125, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.9559026956558228, "rewards/meter/std": 0.015433711931109428, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9444444179534912, "rewards/repeat_penalty/std": 0.059391383081674576, "rewards/total_composite/mean": 0.9024706482887268, "rewards/total_composite/std": 0.052703484892845154, "reward": 0.9024706482887268, "reward_std": 0.05270349606871605, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05570008605718613, "sampling/sampling_logp_difference/max": 2.0624022483825684, "sampling/importance_sampling_ratio/min": 0.12714816629886627, "sampling/importance_sampling_ratio/mean": 1.0074241161346436, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2725033536553383, "clip_ratio/low_mean": 0.01586226187646389, "clip_ratio/low_min": 0.01586226187646389, "clip_ratio/high_mean": 0.020418287720531225, "clip_ratio/high_max": 0.020418287720531225, "clip_ratio/region_mean": 0.036280549596995115, "reward_total_mean": 0.9024706482887268, "reward_meter_mean": 0.9559026956558228, "reward_meter_std": 0.015433711931109428, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9444444179534912, "reward_repeat_penalty_std": 0.059391383081674576, "reward_total_composite_mean": 0.9024706482887268, "reward_total_composite_std": 0.052703484892845154} {"timestamp_utc": "2026-04-12T03:24:47Z", "mode": "train", "global_step": 3135, "epoch": 0.12591878539583082, "loss": 0.0022, "grad_norm": 1.680987000465393, "learning_rate": 5.03030303030303e-07, "num_tokens": 7140487.0, "completions/mean_length": 100.75, "completions/min_length": 100.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.75, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.9992322325706482, "rewards/meter/std": 0.00010666978050721809, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992322325706482, "rewards/total_composite/std": 0.00010666978050721809, "reward": 0.9992322325706482, "reward_std": 0.00010666977323126048, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.029712356626987457, "sampling/sampling_logp_difference/max": 1.3842487335205078, "sampling/importance_sampling_ratio/min": 0.25051194429397583, "sampling/importance_sampling_ratio/mean": 1.006977915763855, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.16831262316554785, "clip_ratio/low_mean": 0.011116024106740952, "clip_ratio/low_min": 0.011116024106740952, "clip_ratio/high_mean": 0.006212871172465384, "clip_ratio/high_max": 0.006212871172465384, "clip_ratio/region_mean": 0.017328895279206336, "reward_total_mean": 0.9992322325706482, "reward_meter_mean": 0.9992322325706482, "reward_meter_std": 0.00010666978050721809, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992322325706482, "reward_total_composite_std": 0.00010666978050721809} {"timestamp_utc": "2026-04-12T03:24:52Z", "mode": "train", "global_step": 3136, "epoch": 0.12595895087761577, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 5.000000000000001e-07, "num_tokens": 7142015.0, "completions/mean_length": 32.0, "completions/min_length": 32.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "reward": 0.9992982149124146, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 7.456184539478272e-05, "sampling/sampling_logp_difference/max": 0.0015591848641633987, "sampling/importance_sampling_ratio/min": 0.999688982963562, "sampling/importance_sampling_ratio/mean": 1.0000699758529663, "sampling/importance_sampling_ratio/max": 1.0015604496002197, "entropy": 0.0006697150638501626, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9992982149124146, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:24:57Z", "mode": "train", "global_step": 3137, "epoch": 0.12599911635940073, "loss": 0.0016, "grad_norm": 1.265021800994873, "learning_rate": 4.96969696969697e-07, "num_tokens": 7143751.0, "completions/mean_length": 68.0, "completions/min_length": 68.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.0, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9994433522224426, "rewards/meter/std": 8.680017344886437e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994433522224426, "rewards/total_composite/std": 8.680017344886437e-05, "reward": 0.9994433522224426, "reward_std": 8.68034694576636e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01834726519882679, "sampling/sampling_logp_difference/max": 1.1407603025436401, "sampling/importance_sampling_ratio/min": 0.4000520408153534, "sampling/importance_sampling_ratio/mean": 1.0005712509155273, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.058580921962857246, "clip_ratio/low_mean": 0.0036764706019312143, "clip_ratio/low_min": 0.0036764706019312143, "clip_ratio/high_mean": 0.011029411805793643, "clip_ratio/high_max": 0.011029411805793643, "clip_ratio/region_mean": 0.014705882407724857, "reward_total_mean": 0.9994433522224426, "reward_meter_mean": 0.9994433522224426, "reward_meter_std": 8.680017344886437e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994433522224426, "reward_total_composite_std": 8.680017344886437e-05} {"timestamp_utc": "2026-04-12T03:25:01Z", "mode": "train", "global_step": 3138, "epoch": 0.12603928184118568, "loss": -0.0077, "grad_norm": 2.8887453079223633, "learning_rate": 4.93939393939394e-07, "num_tokens": 7145214.0, "completions/mean_length": 40.875, "completions/min_length": 40.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.875, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9984819889068604, "rewards/meter/std": 0.0004429934488143772, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9984819889068604, "rewards/total_composite/std": 0.0004429934488143772, "reward": 0.9984819889068604, "reward_std": 0.00044299106230027974, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012076342478394508, "sampling/sampling_logp_difference/max": 0.41540956497192383, "sampling/importance_sampling_ratio/min": 0.6600699424743652, "sampling/importance_sampling_ratio/mean": 1.0034947395324707, "sampling/importance_sampling_ratio/max": 1.4185824394226074, "entropy": 0.12436437420547009, "clip_ratio/low_mean": 0.0030487803742289543, "clip_ratio/low_min": 0.0030487803742289543, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0030487803742289543, "reward_total_mean": 0.9984819889068604, "reward_meter_mean": 0.9984819889068604, "reward_meter_std": 0.0004429934488143772, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9984819889068604, "reward_total_composite_std": 0.0004429934488143772} {"timestamp_utc": "2026-04-12T03:25:06Z", "mode": "train", "global_step": 3139, "epoch": 0.12607944732297063, "loss": -0.0001, "grad_norm": 0.1556718647480011, "learning_rate": 4.909090909090909e-07, "num_tokens": 7146949.0, "completions/mean_length": 66.875, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.875, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981504678726196, "rewards/meter/std": 9.697491805127356e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981504678726196, "rewards/total_composite/std": 9.697491805127356e-06, "reward": 0.9981504678726196, "reward_std": 9.708711331768427e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007338312920182943, "sampling/sampling_logp_difference/max": 0.5803370475769043, "sampling/importance_sampling_ratio/min": 0.5597096681594849, "sampling/importance_sampling_ratio/mean": 1.0014523267745972, "sampling/importance_sampling_ratio/max": 1.3643027544021606, "entropy": 0.04900990845635533, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/high_mean": 0.0037313431967049837, "clip_ratio/high_max": 0.0037313431967049837, "clip_ratio/region_mean": 0.005597014795057476, "reward_total_mean": 0.9981504678726196, "reward_meter_mean": 0.9981504678726196, "reward_meter_std": 9.697491805127356e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981504678726196, "reward_total_composite_std": 9.697491805127356e-06} {"timestamp_utc": "2026-04-12T03:25:14Z", "mode": "train", "global_step": 3140, "epoch": 0.1261196128047556, "loss": -0.0074, "grad_norm": 2.1409451961517334, "learning_rate": 4.878787878787879e-07, "num_tokens": 7151106.0, "completions/mean_length": 311.625, "completions/min_length": 305.0, "completions/max_length": 322.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 311.625, "completions/min_terminated_length": 305.0, "completions/max_terminated_length": 322.0, "rewards/meter/mean": 0.9965137243270874, "rewards/meter/std": 0.00763244554400444, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9965137243270874, "rewards/total_composite/std": 0.00763244554400444, "reward": 0.9965137243270874, "reward_std": 0.0076324427500367165, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0488644503057003, "sampling/sampling_logp_difference/max": 1.3517093658447266, "sampling/importance_sampling_ratio/min": 0.2587974965572357, "sampling/importance_sampling_ratio/mean": 1.0111055374145508, "sampling/importance_sampling_ratio/max": 1.9414808750152588, "entropy": 0.447547797113657, "clip_ratio/low_mean": 0.005737704690545797, "clip_ratio/low_min": 0.005737704690545797, "clip_ratio/high_mean": 0.02794860501307994, "clip_ratio/high_max": 0.02794860501307994, "clip_ratio/region_mean": 0.03368630970362574, "reward_total_mean": 0.9965137243270874, "reward_meter_mean": 0.9965137243270874, "reward_meter_std": 0.00763244554400444, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9965137243270874, "reward_total_composite_std": 0.00763244554400444} {"timestamp_utc": "2026-04-12T03:25:18Z", "mode": "train", "global_step": 3141, "epoch": 0.12615977828654054, "loss": -0.0163, "grad_norm": 5.089842319488525, "learning_rate": 4.848484848484849e-07, "num_tokens": 7152584.0, "completions/mean_length": 33.75, "completions/min_length": 32.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.75, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9182018041610718, "rewards/meter/std": 0.2099631130695343, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9182018041610718, "rewards/total_composite/std": 0.2099631130695343, "reward": 0.9182018041610718, "reward_std": 0.2099630981683731, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012079699896275997, "sampling/sampling_logp_difference/max": 0.8862104415893555, "sampling/importance_sampling_ratio/min": 0.41221490502357483, "sampling/importance_sampling_ratio/mean": 1.0063190460205078, "sampling/importance_sampling_ratio/max": 1.7678229808807373, "entropy": 0.10679818876087666, "clip_ratio/low_mean": 0.0078125, "clip_ratio/low_min": 0.0078125, "clip_ratio/high_mean": 0.0036764706019312143, "clip_ratio/high_max": 0.0036764706019312143, "clip_ratio/region_mean": 0.011488970601931214, "reward_total_mean": 0.9182018041610718, "reward_meter_mean": 0.9182018041610718, "reward_meter_std": 0.2099631130695343, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9182018041610718, "reward_total_composite_std": 0.2099631130695343} {"timestamp_utc": "2026-04-12T03:25:22Z", "mode": "train", "global_step": 3142, "epoch": 0.1261999437683255, "loss": -0.0134, "grad_norm": 3.1886298656463623, "learning_rate": 4.818181818181818e-07, "num_tokens": 7154301.0, "completions/mean_length": 70.625, "completions/min_length": 68.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.625, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9989311099052429, "rewards/meter/std": 0.0014027439756318927, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989311099052429, "rewards/total_composite/std": 0.0014027439756318927, "reward": 0.9989311099052429, "reward_std": 0.0014027438592165709, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013442294672131538, "sampling/sampling_logp_difference/max": 0.9266881942749023, "sampling/importance_sampling_ratio/min": 0.3958625793457031, "sampling/importance_sampling_ratio/mean": 1.0028244256973267, "sampling/importance_sampling_ratio/max": 1.341347575187683, "entropy": 0.09765590541064739, "clip_ratio/low_mean": 0.0036764706019312143, "clip_ratio/low_min": 0.0036764706019312143, "clip_ratio/high_mean": 0.0035211266949772835, "clip_ratio/high_max": 0.0035211266949772835, "clip_ratio/region_mean": 0.007197597296908498, "reward_total_mean": 0.9989311099052429, "reward_meter_mean": 0.9989311099052429, "reward_meter_std": 0.0014027439756318927, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9989311099052429, "reward_total_composite_std": 0.0014027439756318927} {"timestamp_utc": "2026-04-12T03:25:28Z", "mode": "train", "global_step": 3143, "epoch": 0.12624010925011045, "loss": 0.0034, "grad_norm": 1.7062424421310425, "learning_rate": 4.787878787878789e-07, "num_tokens": 7156840.0, "completions/mean_length": 133.375, "completions/min_length": 128.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 133.375, "completions/min_terminated_length": 128.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.9990321397781372, "rewards/meter/std": 0.00018018556875176728, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990321397781372, "rewards/total_composite/std": 0.00018018556875176728, "reward": 0.9990321397781372, "reward_std": 0.00018020268180407584, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.035543572157621384, "sampling/sampling_logp_difference/max": 1.157923698425293, "sampling/importance_sampling_ratio/min": 0.3141377568244934, "sampling/importance_sampling_ratio/mean": 1.0070942640304565, "sampling/importance_sampling_ratio/max": 1.9823683500289917, "entropy": 0.29481677152216434, "clip_ratio/low_mean": 0.013188510201871395, "clip_ratio/low_min": 0.013188510201871395, "clip_ratio/high_mean": 0.01591338007710874, "clip_ratio/high_max": 0.01591338007710874, "clip_ratio/region_mean": 0.029101890278980136, "reward_total_mean": 0.9990321397781372, "reward_meter_mean": 0.9990321397781372, "reward_meter_std": 0.00018018556875176728, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990321397781372, "reward_total_composite_std": 0.00018018556875176728} {"timestamp_utc": "2026-04-12T03:25:32Z", "mode": "train", "global_step": 3144, "epoch": 0.1262802747318954, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.757575757575758e-07, "num_tokens": 7158856.0, "completions/mean_length": 80.0, "completions/min_length": 80.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.0, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "reward": 0.7190229296684265, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00035932144965045154, "sampling/sampling_logp_difference/max": 0.012860830873250961, "sampling/importance_sampling_ratio/min": 0.9996905326843262, "sampling/importance_sampling_ratio/mean": 1.0003576278686523, "sampling/importance_sampling_ratio/max": 1.0129438638687134, "entropy": 0.0032840893836691976, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7190229296684265, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:25:37Z", "mode": "train", "global_step": 3145, "epoch": 0.12632044021368036, "loss": 0.0038, "grad_norm": 3.3756773471832275, "learning_rate": 4.7272727272727273e-07, "num_tokens": 7161044.0, "completions/mean_length": 96.5, "completions/min_length": 95.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.5, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9917601346969604, "rewards/meter/std": 0.004016265273094177, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9917601346969604, "rewards/total_composite/std": 0.004016265273094177, "reward": 0.9917601346969604, "reward_std": 0.004016268998384476, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031790200620889664, "sampling/sampling_logp_difference/max": 1.1975455284118652, "sampling/importance_sampling_ratio/min": 0.3019343912601471, "sampling/importance_sampling_ratio/mean": 1.0063228607177734, "sampling/importance_sampling_ratio/max": 1.6533253192901611, "entropy": 0.23462153412401676, "clip_ratio/low_mean": 0.014030612306669354, "clip_ratio/low_min": 0.014030612306669354, "clip_ratio/high_mean": 0.014405153575353324, "clip_ratio/high_max": 0.014405153575353324, "clip_ratio/region_mean": 0.02843576588202268, "reward_total_mean": 0.9917601346969604, "reward_meter_mean": 0.9917601346969604, "reward_meter_std": 0.004016265273094177, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9917601346969604, "reward_total_composite_std": 0.004016265273094177} {"timestamp_utc": "2026-04-12T03:25:41Z", "mode": "train", "global_step": 3146, "epoch": 0.1263606056954653, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.696969696969697e-07, "num_tokens": 7162452.0, "completions/mean_length": 29.0, "completions/min_length": 29.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9957436919212341, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957436919212341, "rewards/total_composite/std": 0.0, "reward": 0.9957436919212341, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00029772642301395535, "sampling/sampling_logp_difference/max": 0.008134791627526283, "sampling/importance_sampling_ratio/min": 0.9950645565986633, "sampling/importance_sampling_ratio/mean": 1.0002386569976807, "sampling/importance_sampling_ratio/max": 1.0081679821014404, "entropy": 0.00371554572484456, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9957436919212341, "reward_meter_mean": 0.9957436919212341, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9957436919212341, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:25:49Z", "mode": "train", "global_step": 3147, "epoch": 0.12640077117725027, "loss": 0.0288, "grad_norm": 1.647410273551941, "learning_rate": 4.666666666666667e-07, "num_tokens": 7166497.0, "completions/mean_length": 315.625, "completions/min_length": 309.0, "completions/max_length": 339.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 315.625, "completions/min_terminated_length": 309.0, "completions/max_terminated_length": 339.0, "rewards/meter/mean": 0.9990444779396057, "rewards/meter/std": 0.00020862040400970727, "rewards/count_adherence/mean": 0.984375, "rewards/count_adherence/std": 0.04419417306780815, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9843137264251709, "rewards/repeat_penalty/std": 0.029120875522494316, "rewards/total_composite/mean": 0.9686882495880127, "rewards/total_composite/std": 0.06351150572299957, "reward": 0.9686882495880127, "reward_std": 0.06351151317358017, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04411856085062027, "sampling/sampling_logp_difference/max": 1.3515138626098633, "sampling/importance_sampling_ratio/min": 0.25884810090065, "sampling/importance_sampling_ratio/mean": 1.0113532543182373, "sampling/importance_sampling_ratio/max": 1.7542108297348022, "entropy": 0.41412847116589546, "clip_ratio/low_mean": 0.004168422543443739, "clip_ratio/low_min": 0.004168422543443739, "clip_ratio/high_mean": 0.017632243572734296, "clip_ratio/high_max": 0.017632243572734296, "clip_ratio/region_mean": 0.021800666116178036, "reward_total_mean": 0.9686882495880127, "reward_meter_mean": 0.9990444779396057, "reward_meter_std": 0.00020862040400970727, "reward_count_adherence_mean": 0.984375, "reward_count_adherence_std": 0.04419417306780815, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9843137264251709, "reward_repeat_penalty_std": 0.029120875522494316, "reward_total_composite_mean": 0.9686882495880127, "reward_total_composite_std": 0.06351150572299957} {"timestamp_utc": "2026-04-12T03:25:54Z", "mode": "train", "global_step": 3148, "epoch": 0.12644093665903522, "loss": 0.0009, "grad_norm": 0.7614604234695435, "learning_rate": 4.6363636363636365e-07, "num_tokens": 7168719.0, "completions/mean_length": 106.75, "completions/min_length": 106.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.75, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.9992802143096924, "rewards/meter/std": 0.0001533372706035152, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992802143096924, "rewards/total_composite/std": 0.0001533372706035152, "reward": 0.9992802143096924, "reward_std": 0.00015333489864133298, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013726499862968922, "sampling/sampling_logp_difference/max": 1.1793193817138672, "sampling/importance_sampling_ratio/min": 0.30748796463012695, "sampling/importance_sampling_ratio/mean": 1.0032622814178467, "sampling/importance_sampling_ratio/max": 1.5043672323226929, "entropy": 0.14908354356884956, "clip_ratio/low_mean": 0.008188590873032808, "clip_ratio/low_min": 0.008188590873032808, "clip_ratio/high_mean": 0.00821063295006752, "clip_ratio/high_max": 0.00821063295006752, "clip_ratio/region_mean": 0.01639922382310033, "reward_total_mean": 0.9992802143096924, "reward_meter_mean": 0.9992802143096924, "reward_meter_std": 0.0001533372706035152, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992802143096924, "reward_total_composite_std": 0.0001533372706035152} {"timestamp_utc": "2026-04-12T03:26:04Z", "mode": "train", "global_step": 3149, "epoch": 0.12648110214082017, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.6060606060606064e-07, "num_tokens": 7170575.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9272494316101074, "rewards/meter/std": 0.10411977022886276, "rewards/count_adherence/mean": 0.6687499284744263, "rewards/count_adherence/std": 0.025877466425299644, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9813033938407898, "rewards/repeat_penalty/std": 0.03969252109527588, "rewards/total_composite/mean": 0.6074195504188538, "rewards/total_composite/std": 0.06437568366527557, "reward": 0.6074195504188538, "reward_std": 0.06437568366527557, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.6074195504188538, "reward_meter_mean": 0.9272494316101074, "reward_meter_std": 0.10411977022886276, "reward_count_adherence_mean": 0.6687499284744263, "reward_count_adherence_std": 0.025877466425299644, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9813033938407898, "reward_repeat_penalty_std": 0.03969252109527588, "reward_total_composite_mean": 0.6074195504188538, "reward_total_composite_std": 0.06437568366527557} {"timestamp_utc": "2026-04-12T03:26:08Z", "mode": "train", "global_step": 3150, "epoch": 0.12652126762260513, "loss": 0.0003, "grad_norm": 0.012665612623095512, "learning_rate": 4.5757575757575764e-07, "num_tokens": 7172439.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.997339129447937, "rewards/meter/std": 3.875939285080676e-07, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.997339129447937, "rewards/total_composite/std": 3.875939285080676e-07, "reward": 0.997339129447937, "reward_std": 3.8628223819614504e-07, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004391301888972521, "sampling/sampling_logp_difference/max": 0.7720130681991577, "sampling/importance_sampling_ratio/min": 0.4620819389820099, "sampling/importance_sampling_ratio/mean": 0.9999790787696838, "sampling/importance_sampling_ratio/max": 1.506388783454895, "entropy": 0.021077871089801192, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0020491802133619785, "clip_ratio/high_max": 0.0020491802133619785, "clip_ratio/region_mean": 0.0020491802133619785, "reward_total_mean": 0.997339129447937, "reward_meter_mean": 0.997339129447937, "reward_meter_std": 3.875939285080676e-07, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.997339129447937, "reward_total_composite_std": 3.875939285080676e-07} {"timestamp_utc": "2026-04-12T03:27:29Z", "mode": "eval", "global_step": 3150, "epoch": 0.12652126762260513, "eval_loss": NaN, "eval_runtime": 80.4009, "eval_samples_per_second": 1.294, "eval_steps_per_second": 0.162, "eval_num_tokens": 7172439.0, "eval_completions/mean_length": 217.70192307692307, "eval_completions/min_length": 61.69230769230769, "eval_completions/max_length": 436.2307692307692, "eval_completions/clipped_ratio": 0.038461538461538464, "eval_completions/mean_terminated_length": 206.14835533728967, "eval_completions/min_terminated_length": 61.69230769230769, "eval_completions/max_terminated_length": 401.84615384615387, "eval_rewards/meter/mean": 0.7973346756054804, "eval_rewards/meter/std": 0.3138508295210508, "eval_rewards/count_adherence/mean": 0.963410904774299, "eval_rewards/count_adherence/std": 0.054318822060640044, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/repeat_penalty/mean": 0.9485706686973572, "eval_rewards/repeat_penalty/std": 0.08086307346820831, "eval_rewards/total_composite/mean": 0.7355112204184899, "eval_rewards/total_composite/std": 0.31246042595459866, "eval_reward": 0.7355112204184899, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.0324688284442975, "eval_sampling/sampling_logp_difference/max": 1.1065230736365685, "eval_sampling/importance_sampling_ratio/min": 0.33571984332341415, "eval_sampling/importance_sampling_ratio/mean": 1.0091725221047034, "eval_sampling/importance_sampling_ratio/max": 1.4886829761358409, "eval_entropy": 0.37681426910253674, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7355112204184899, "eval_reward_meter_mean": 0.7973346756054804, "eval_reward_meter_std": 0.3138508295210508, "eval_reward_count_adherence_mean": 0.963410904774299, "eval_reward_count_adherence_std": 0.054318822060640044, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_repeat_penalty_mean": 0.9485706686973572, "eval_reward_repeat_penalty_std": 0.08086307346820831, "eval_reward_total_composite_mean": 0.7355112204184899, "eval_reward_total_composite_std": 0.31246042595459866} {"timestamp_utc": "2026-04-12T03:27:37Z", "mode": "train", "global_step": 3151, "epoch": 0.12656143310439008, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.5454545454545457e-07, "num_tokens": 7174111.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00020680490706581622, "sampling/sampling_logp_difference/max": 0.002479560673236847, "sampling/importance_sampling_ratio/min": 0.9986385703086853, "sampling/importance_sampling_ratio/mean": 1.000199317932129, "sampling/importance_sampling_ratio/max": 1.0024826526641846, "entropy": 0.0016731252981116995, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:27:41Z", "mode": "train", "global_step": 3152, "epoch": 0.12660159858617503, "loss": 0.0014, "grad_norm": 1.8619328737258911, "learning_rate": 4.5151515151515156e-07, "num_tokens": 7175864.0, "completions/mean_length": 65.125, "completions/min_length": 65.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.125, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9934186935424805, "rewards/meter/std": 0.00011813976016128436, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9934186935424805, "rewards/total_composite/std": 0.00011813976016128436, "reward": 0.9934186935424805, "reward_std": 0.00011814001481980085, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010398104786872864, "sampling/sampling_logp_difference/max": 0.37988075613975525, "sampling/importance_sampling_ratio/min": 0.683942973613739, "sampling/importance_sampling_ratio/mean": 1.0027790069580078, "sampling/importance_sampling_ratio/max": 1.3604105710983276, "entropy": 0.06057531526312232, "clip_ratio/low_mean": 0.007634032750502229, "clip_ratio/low_min": 0.007634032750502229, "clip_ratio/high_mean": 0.003846153849735856, "clip_ratio/high_max": 0.003846153849735856, "clip_ratio/region_mean": 0.011480186600238085, "reward_total_mean": 0.9934186935424805, "reward_meter_mean": 0.9934186935424805, "reward_meter_std": 0.00011813976016128436, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9934186935424805, "reward_total_composite_std": 0.00011813976016128436} {"timestamp_utc": "2026-04-12T03:27:48Z", "mode": "train", "global_step": 3153, "epoch": 0.12664176406796, "loss": 0.0065, "grad_norm": 2.150918483734131, "learning_rate": 4.484848484848485e-07, "num_tokens": 7179187.0, "completions/mean_length": 199.375, "completions/min_length": 197.0, "completions/max_length": 204.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 199.375, "completions/min_terminated_length": 197.0, "completions/max_terminated_length": 204.0, "rewards/meter/mean": 0.9993048906326294, "rewards/meter/std": 0.0001919995847856626, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9318181276321411, "rewards/repeat_penalty/std": 0.08058229833841324, "rewards/total_composite/mean": 0.9311637878417969, "rewards/total_composite/std": 0.08044037967920303, "reward": 0.9311637878417969, "reward_std": 0.08044035732746124, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020360471680760384, "sampling/sampling_logp_difference/max": 1.146202564239502, "sampling/importance_sampling_ratio/min": 0.31784144043922424, "sampling/importance_sampling_ratio/mean": 1.0063588619232178, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.18566425889730453, "clip_ratio/low_mean": 0.0037722690613009036, "clip_ratio/low_min": 0.0037722690613009036, "clip_ratio/high_mean": 0.006870172452181578, "clip_ratio/high_max": 0.006870172452181578, "clip_ratio/region_mean": 0.010642441513482481, "reward_total_mean": 0.9311637878417969, "reward_meter_mean": 0.9993048906326294, "reward_meter_std": 0.0001919995847856626, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9318181276321411, "reward_repeat_penalty_std": 0.08058229833841324, "reward_total_composite_mean": 0.9311637878417969, "reward_total_composite_std": 0.08044037967920303} {"timestamp_utc": "2026-04-12T03:27:53Z", "mode": "train", "global_step": 3154, "epoch": 0.12668192954974494, "loss": 0.0047, "grad_norm": 3.9876134395599365, "learning_rate": 4.454545454545455e-07, "num_tokens": 7181304.0, "completions/mean_length": 103.625, "completions/min_length": 102.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.625, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.9898449182510376, "rewards/meter/std": 0.012184920720756054, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9898449182510376, "rewards/total_composite/std": 0.012184920720756054, "reward": 0.9898449182510376, "reward_std": 0.01218490768224001, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.036951709538698196, "sampling/sampling_logp_difference/max": 1.5696239471435547, "sampling/importance_sampling_ratio/min": 0.20812343060970306, "sampling/importance_sampling_ratio/mean": 1.005436658859253, "sampling/importance_sampling_ratio/max": 1.9000133275985718, "entropy": 0.30003978684544563, "clip_ratio/low_mean": 0.007282239850610495, "clip_ratio/low_min": 0.007282239850610495, "clip_ratio/high_mean": 0.021670067915692925, "clip_ratio/high_max": 0.021670067915692925, "clip_ratio/region_mean": 0.02895230776630342, "reward_total_mean": 0.9898449182510376, "reward_meter_mean": 0.9898449182510376, "reward_meter_std": 0.012184920720756054, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9898449182510376, "reward_total_composite_std": 0.012184920720756054} {"timestamp_utc": "2026-04-12T03:27:57Z", "mode": "train", "global_step": 3155, "epoch": 0.1267220950315299, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.424242424242425e-07, "num_tokens": 7183088.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9973388910293579, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973388910293579, "rewards/total_composite/std": 0.0, "reward": 0.9973388910293579, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0017873203614726663, "sampling/sampling_logp_difference/max": 0.15021102130413055, "sampling/importance_sampling_ratio/min": 0.8605263233184814, "sampling/importance_sampling_ratio/mean": 1.0003093481063843, "sampling/importance_sampling_ratio/max": 1.055625557899475, "entropy": 0.013011510367505252, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9973388910293579, "reward_meter_mean": 0.9973388910293579, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9973388910293579, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:28:02Z", "mode": "train", "global_step": 3156, "epoch": 0.12676226051331485, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.393939393939394e-07, "num_tokens": 7184864.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002461693366058171, "sampling/sampling_logp_difference/max": 0.0035922862589359283, "sampling/importance_sampling_ratio/min": 0.9997955560684204, "sampling/importance_sampling_ratio/mean": 1.000244379043579, "sampling/importance_sampling_ratio/max": 1.0035988092422485, "entropy": 0.0018204424268333241, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:28:06Z", "mode": "train", "global_step": 3157, "epoch": 0.1268024259950998, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.363636363636364e-07, "num_tokens": 7186584.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00020154265803284943, "sampling/sampling_logp_difference/max": 0.002117213560268283, "sampling/importance_sampling_ratio/min": 0.9994971752166748, "sampling/importance_sampling_ratio/mean": 1.0001963376998901, "sampling/importance_sampling_ratio/max": 1.0021194219589233, "entropy": 0.0016821113676996902, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:28:11Z", "mode": "train", "global_step": 3158, "epoch": 0.12684259147688476, "loss": 0.0133, "grad_norm": 4.2719950675964355, "learning_rate": 4.333333333333334e-07, "num_tokens": 7188342.0, "completions/mean_length": 68.75, "completions/min_length": 66.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.75, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9841428399085999, "rewards/meter/std": 0.03312673419713974, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9841428399085999, "rewards/total_composite/std": 0.03312673419713974, "reward": 0.9841428399085999, "reward_std": 0.03312673047184944, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03139575570821762, "sampling/sampling_logp_difference/max": 1.0852737426757812, "sampling/importance_sampling_ratio/min": 0.337809294462204, "sampling/importance_sampling_ratio/mean": 1.007124900817871, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23758242093026638, "clip_ratio/low_mean": 0.0035211266949772835, "clip_ratio/low_min": 0.0035211266949772835, "clip_ratio/high_mean": 0.009170901728793979, "clip_ratio/high_max": 0.009170901728793979, "clip_ratio/region_mean": 0.012692028423771262, "reward_total_mean": 0.9841428399085999, "reward_meter_mean": 0.9841428399085999, "reward_meter_std": 0.03312673419713974, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9841428399085999, "reward_total_composite_std": 0.03312673419713974} {"timestamp_utc": "2026-04-12T03:28:15Z", "mode": "train", "global_step": 3159, "epoch": 0.1268827569586697, "loss": 0.0047, "grad_norm": 6.7398481369018555, "learning_rate": 4.3030303030303034e-07, "num_tokens": 7190330.0, "completions/mean_length": 86.5, "completions/min_length": 83.0, "completions/max_length": 89.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.5, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 89.0, "rewards/meter/mean": 0.8662310242652893, "rewards/meter/std": 0.18711434304714203, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.8583756685256958, "rewards/total_composite/std": 0.20775045454502106, "reward": 0.8583756685256958, "reward_std": 0.20775045454502106, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05684678256511688, "sampling/sampling_logp_difference/max": 1.4105424880981445, "sampling/importance_sampling_ratio/min": 0.3277885615825653, "sampling/importance_sampling_ratio/mean": 1.0018366575241089, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.24941212497651577, "clip_ratio/low_mean": 0.01434953324496746, "clip_ratio/low_min": 0.01434953324496746, "clip_ratio/high_mean": 0.04191699391230941, "clip_ratio/high_max": 0.04191699391230941, "clip_ratio/region_mean": 0.05626652715727687, "reward_total_mean": 0.8583756685256958, "reward_meter_mean": 0.8662310242652893, "reward_meter_std": 0.18711434304714203, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.8583756685256958, "reward_total_composite_std": 0.20775045454502106} {"timestamp_utc": "2026-04-12T03:28:25Z", "mode": "train", "global_step": 3160, "epoch": 0.12692292244045467, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.272727272727273e-07, "num_tokens": 7192010.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9984534978866577, "rewards/meter/std": 0.0011360765201970935, "rewards/count_adherence/mean": 0.6624999642372131, "rewards/count_adherence/std": 0.023145509883761406, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.9809472560882568, "rewards/repeat_penalty/std": 0.020373545587062836, "rewards/total_composite/mean": 0.5677040219306946, "rewards/total_composite/std": 0.2308495044708252, "reward": 0.5677040219306946, "reward_std": 0.230849489569664, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.5677040219306946, "reward_meter_mean": 0.9984534978866577, "reward_meter_std": 0.0011360765201970935, "reward_count_adherence_mean": 0.6624999642372131, "reward_count_adherence_std": 0.023145509883761406, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.9809472560882568, "reward_repeat_penalty_std": 0.020373545587062836, "reward_total_composite_mean": 0.5677040219306946, "reward_total_composite_std": 0.2308495044708252} {"timestamp_utc": "2026-04-12T03:28:34Z", "mode": "train", "global_step": 3161, "epoch": 0.12696308792223962, "loss": -0.0163, "grad_norm": 1.4183757305145264, "learning_rate": 4.242424242424243e-07, "num_tokens": 7196839.0, "completions/mean_length": 349.625, "completions/min_length": 338.0, "completions/max_length": 359.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 349.625, "completions/min_terminated_length": 338.0, "completions/max_terminated_length": 359.0, "rewards/meter/mean": 0.9987608194351196, "rewards/meter/std": 0.00043402795563451946, "rewards/count_adherence/mean": 0.8020833134651184, "rewards/count_adherence/std": 0.0431290864944458, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9601608514785767, "rewards/repeat_penalty/std": 0.02460995875298977, "rewards/total_composite/mean": 0.7685060501098633, "rewards/total_composite/std": 0.03033612295985222, "reward": 0.7685060501098633, "reward_std": 0.030336128547787666, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.034961603581905365, "sampling/sampling_logp_difference/max": 1.671947956085205, "sampling/importance_sampling_ratio/min": 0.1878807246685028, "sampling/importance_sampling_ratio/mean": 1.0060912370681763, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.35866962373256683, "clip_ratio/low_mean": 0.008485189639031887, "clip_ratio/low_min": 0.008485189639031887, "clip_ratio/high_mean": 0.013319394318386912, "clip_ratio/high_max": 0.013319394318386912, "clip_ratio/region_mean": 0.0218045839574188, "reward_total_mean": 0.7685060501098633, "reward_meter_mean": 0.9987608194351196, "reward_meter_std": 0.00043402795563451946, "reward_count_adherence_mean": 0.8020833134651184, "reward_count_adherence_std": 0.0431290864944458, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9601608514785767, "reward_repeat_penalty_std": 0.02460995875298977, "reward_total_composite_mean": 0.7685060501098633, "reward_total_composite_std": 0.03033612295985222} {"timestamp_utc": "2026-04-12T03:28:38Z", "mode": "train", "global_step": 3162, "epoch": 0.12700325340402457, "loss": -0.0002, "grad_norm": 0.1628890037536621, "learning_rate": 4.2121212121212126e-07, "num_tokens": 7198784.0, "completions/mean_length": 71.125, "completions/min_length": 71.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.125, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9994406700134277, "rewards/meter/std": 1.3193358427088242e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994406700134277, "rewards/total_composite/std": 1.3193358427088242e-05, "reward": 0.9994406700134277, "reward_std": 1.3199281056586187e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0068392762914299965, "sampling/sampling_logp_difference/max": 0.3337416648864746, "sampling/importance_sampling_ratio/min": 0.716238796710968, "sampling/importance_sampling_ratio/mean": 1.003661036491394, "sampling/importance_sampling_ratio/max": 1.3171619176864624, "entropy": 0.07914929743856192, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0035211266949772835, "clip_ratio/high_max": 0.0035211266949772835, "clip_ratio/region_mean": 0.0035211266949772835, "reward_total_mean": 0.9994406700134277, "reward_meter_mean": 0.9994406700134277, "reward_meter_std": 1.3193358427088242e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994406700134277, "reward_total_composite_std": 1.3193358427088242e-05} {"timestamp_utc": "2026-04-12T03:28:43Z", "mode": "train", "global_step": 3163, "epoch": 0.12704341888580953, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.181818181818182e-07, "num_tokens": 7200688.0, "completions/mean_length": 80.0, "completions/min_length": 80.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.0, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "reward": 0.7190229296684265, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0004915767349302769, "sampling/sampling_logp_difference/max": 0.022466091439127922, "sampling/importance_sampling_ratio/min": 0.9993630647659302, "sampling/importance_sampling_ratio/mean": 1.0004879236221313, "sampling/importance_sampling_ratio/max": 1.0227203369140625, "entropy": 0.003859572723740712, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7190229296684265, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:28:47Z", "mode": "train", "global_step": 3164, "epoch": 0.12708358436759448, "loss": -0.0002, "grad_norm": 7.426182270050049, "learning_rate": 4.1515151515151513e-07, "num_tokens": 7202591.0, "completions/mean_length": 45.875, "completions/min_length": 45.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.875, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.9413228631019592, "rewards/meter/std": 0.004475572612136602, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9413228631019592, "rewards/total_composite/std": 0.004475572612136602, "reward": 0.9413228631019592, "reward_std": 0.004475583788007498, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.033416956663131714, "sampling/sampling_logp_difference/max": 2.256293296813965, "sampling/importance_sampling_ratio/min": 0.10473800450563431, "sampling/importance_sampling_ratio/mean": 0.9981755614280701, "sampling/importance_sampling_ratio/max": 1.8759413957595825, "entropy": 0.10774505604058504, "clip_ratio/low_mean": 0.0027173913549631834, "clip_ratio/low_min": 0.0027173913549631834, "clip_ratio/high_mean": 0.019021739484742284, "clip_ratio/high_max": 0.019021739484742284, "clip_ratio/region_mean": 0.021739130839705467, "reward_total_mean": 0.9413228631019592, "reward_meter_mean": 0.9413228631019592, "reward_meter_std": 0.004475572612136602, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9413228631019592, "reward_total_composite_std": 0.004475572612136602} {"timestamp_utc": "2026-04-12T03:28:52Z", "mode": "train", "global_step": 3165, "epoch": 0.12712374984937944, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.121212121212122e-07, "num_tokens": 7204447.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00010137181379832327, "sampling/sampling_logp_difference/max": 0.0022658593952655792, "sampling/importance_sampling_ratio/min": 0.9977367520332336, "sampling/importance_sampling_ratio/mean": 1.000064730644226, "sampling/importance_sampling_ratio/max": 1.0019134283065796, "entropy": 0.0008098809266812168, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:28:56Z", "mode": "train", "global_step": 3166, "epoch": 0.1271639153311644, "loss": -0.0001, "grad_norm": 0.4401102066040039, "learning_rate": 4.090909090909091e-07, "num_tokens": 7206119.0, "completions/mean_length": 67.0, "completions/min_length": 67.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981397390365601, "rewards/meter/std": 1.352704748569522e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981397390365601, "rewards/total_composite/std": 1.352704748569522e-05, "reward": 0.9981397390365601, "reward_std": 1.3527009286917746e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006174927577376366, "sampling/sampling_logp_difference/max": 0.37103140354156494, "sampling/importance_sampling_ratio/min": 0.6950086951255798, "sampling/importance_sampling_ratio/mean": 1.0013184547424316, "sampling/importance_sampling_ratio/max": 1.4492285251617432, "entropy": 0.052269697189331055, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/high_mean": 0.0018656715983524919, "clip_ratio/high_max": 0.0018656715983524919, "clip_ratio/region_mean": 0.0037313431967049837, "reward_total_mean": 0.9981397390365601, "reward_meter_mean": 0.9981397390365601, "reward_meter_std": 1.352704748569522e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981397390365601, "reward_total_composite_std": 1.352704748569522e-05} {"timestamp_utc": "2026-04-12T03:29:01Z", "mode": "train", "global_step": 3167, "epoch": 0.12720408081294934, "loss": 0.0009, "grad_norm": 1.5397974252700806, "learning_rate": 4.0606060606060605e-07, "num_tokens": 7208456.0, "completions/mean_length": 125.125, "completions/min_length": 124.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 125.125, "completions/min_terminated_length": 124.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.9850419759750366, "rewards/meter/std": 0.024076560512185097, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9673014283180237, "rewards/total_composite/std": 0.052494850009679794, "reward": 0.9673014283180237, "reward_std": 0.0524948425590992, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02755284309387207, "sampling/sampling_logp_difference/max": 1.6096177101135254, "sampling/importance_sampling_ratio/min": 0.19996404647827148, "sampling/importance_sampling_ratio/mean": 1.0092271566390991, "sampling/importance_sampling_ratio/max": 1.744919776916504, "entropy": 0.20761017128825188, "clip_ratio/low_mean": 0.006024193484336138, "clip_ratio/low_min": 0.006024193484336138, "clip_ratio/high_mean": 0.012976191123016179, "clip_ratio/high_max": 0.012976191123016179, "clip_ratio/region_mean": 0.019000384607352316, "reward_total_mean": 0.9673014283180237, "reward_meter_mean": 0.9850419759750366, "reward_meter_std": 0.024076560512185097, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.9673014283180237, "reward_total_composite_std": 0.052494850009679794} {"timestamp_utc": "2026-04-12T03:29:06Z", "mode": "train", "global_step": 3168, "epoch": 0.1272442462947343, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.030303030303031e-07, "num_tokens": 7210280.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 8.32213117973879e-05, "sampling/sampling_logp_difference/max": 0.004544587340205908, "sampling/importance_sampling_ratio/min": 0.9954656958580017, "sampling/importance_sampling_ratio/mean": 1.000028371810913, "sampling/importance_sampling_ratio/max": 1.0013597011566162, "entropy": 0.0007030887718428858, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:29:11Z", "mode": "train", "global_step": 3169, "epoch": 0.12728441177651925, "loss": 0.0016, "grad_norm": 2.7400386333465576, "learning_rate": 4.0000000000000003e-07, "num_tokens": 7212758.0, "completions/mean_length": 137.75, "completions/min_length": 132.0, "completions/max_length": 142.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 137.75, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 142.0, "rewards/meter/mean": 0.9911104440689087, "rewards/meter/std": 0.00600889977067709, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.973473310470581, "rewards/total_composite/std": 0.05160484462976456, "reward": 0.973473310470581, "reward_std": 0.051604848355054855, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04144033044576645, "sampling/sampling_logp_difference/max": 1.1764769554138184, "sampling/importance_sampling_ratio/min": 0.3083631992340088, "sampling/importance_sampling_ratio/mean": 1.0084030628204346, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3814433552324772, "clip_ratio/low_mean": 0.0036231884732842445, "clip_ratio/low_min": 0.0036231884732842445, "clip_ratio/high_mean": 0.028263273648917675, "clip_ratio/high_max": 0.028263273648917675, "clip_ratio/region_mean": 0.03188646212220192, "reward_total_mean": 0.973473310470581, "reward_meter_mean": 0.9911104440689087, "reward_meter_std": 0.00600889977067709, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.973473310470581, "reward_total_composite_std": 0.05160484462976456} {"timestamp_utc": "2026-04-12T03:29:18Z", "mode": "train", "global_step": 3170, "epoch": 0.1273245772583042, "loss": 0.0069, "grad_norm": 1.578904628753662, "learning_rate": 3.9696969696969697e-07, "num_tokens": 7216446.0, "completions/mean_length": 248.0, "completions/min_length": 244.0, "completions/max_length": 251.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 248.0, "completions/min_terminated_length": 244.0, "completions/max_terminated_length": 251.0, "rewards/meter/mean": 0.9988981485366821, "rewards/meter/std": 0.0001572021865285933, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9615384340286255, "rewards/repeat_penalty/std": 0.041117113083601, "rewards/total_composite/mean": 0.9604784250259399, "rewards/total_composite/std": 0.04105609655380249, "reward": 0.9604784250259399, "reward_std": 0.041056081652641296, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027040995657444, "sampling/sampling_logp_difference/max": 1.3942060470581055, "sampling/importance_sampling_ratio/min": 0.248029887676239, "sampling/importance_sampling_ratio/mean": 1.0076178312301636, "sampling/importance_sampling_ratio/max": 1.6617929935455322, "entropy": 0.3082658052444458, "clip_ratio/low_mean": 0.004030561889521778, "clip_ratio/low_min": 0.004030561889521778, "clip_ratio/high_mean": 0.0075775557197630405, "clip_ratio/high_max": 0.0075775557197630405, "clip_ratio/region_mean": 0.011608117609284818, "reward_total_mean": 0.9604784250259399, "reward_meter_mean": 0.9988981485366821, "reward_meter_std": 0.0001572021865285933, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9615384340286255, "reward_repeat_penalty_std": 0.041117113083601, "reward_total_composite_mean": 0.9604784250259399, "reward_total_composite_std": 0.04105609655380249} {"timestamp_utc": "2026-04-12T03:29:23Z", "mode": "train", "global_step": 3171, "epoch": 0.12736474274008916, "loss": -0.0007, "grad_norm": 2.6443123817443848, "learning_rate": 3.9393939393939396e-07, "num_tokens": 7218213.0, "completions/mean_length": 69.875, "completions/min_length": 68.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.875, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9944202899932861, "rewards/meter/std": 0.002229472389444709, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9944202899932861, "rewards/total_composite/std": 0.002229472389444709, "reward": 0.9944202899932861, "reward_std": 0.002229455392807722, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02252027578651905, "sampling/sampling_logp_difference/max": 1.6969127655029297, "sampling/importance_sampling_ratio/min": 0.183248370885849, "sampling/importance_sampling_ratio/mean": 1.0053551197052002, "sampling/importance_sampling_ratio/max": 1.854751706123352, "entropy": 0.1526617044582963, "clip_ratio/low_mean": 0.005335517227649689, "clip_ratio/low_min": 0.005335517227649689, "clip_ratio/high_mean": 0.009086863487027586, "clip_ratio/high_max": 0.009086863487027586, "clip_ratio/region_mean": 0.014422380714677274, "reward_total_mean": 0.9944202899932861, "reward_meter_mean": 0.9944202899932861, "reward_meter_std": 0.002229472389444709, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9944202899932861, "reward_total_composite_std": 0.002229472389444709} {"timestamp_utc": "2026-04-12T03:29:27Z", "mode": "train", "global_step": 3172, "epoch": 0.1274049082218741, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.9090909090909095e-07, "num_tokens": 7219917.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 9.710989979794249e-05, "sampling/sampling_logp_difference/max": 0.0017959647811949253, "sampling/importance_sampling_ratio/min": 0.9985963106155396, "sampling/importance_sampling_ratio/mean": 1.0000759363174438, "sampling/importance_sampling_ratio/max": 1.0017976760864258, "entropy": 0.0011070799009758048, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:29:34Z", "mode": "train", "global_step": 3173, "epoch": 0.12744507370365907, "loss": 0.0057, "grad_norm": 1.6205352544784546, "learning_rate": 3.878787878787879e-07, "num_tokens": 7223459.0, "completions/mean_length": 230.75, "completions/min_length": 226.0, "completions/max_length": 234.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 230.75, "completions/min_terminated_length": 226.0, "completions/max_terminated_length": 234.0, "rewards/meter/mean": 0.9989705085754395, "rewards/meter/std": 0.00055393495131284, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9615384340286255, "rewards/repeat_penalty/std": 0.058148376643657684, "rewards/total_composite/mean": 0.9605352282524109, "rewards/total_composite/std": 0.05785057693719864, "reward": 0.9605352282524109, "reward_std": 0.05785060301423073, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025763755664229393, "sampling/sampling_logp_difference/max": 1.1109724044799805, "sampling/importance_sampling_ratio/min": 0.3292386531829834, "sampling/importance_sampling_ratio/mean": 1.0043230056762695, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2557064164429903, "clip_ratio/low_mean": 0.006500809220597148, "clip_ratio/low_min": 0.006500809220597148, "clip_ratio/high_mean": 0.016817589348647743, "clip_ratio/high_max": 0.016817589348647743, "clip_ratio/region_mean": 0.02331839856924489, "reward_total_mean": 0.9605352282524109, "reward_meter_mean": 0.9989705085754395, "reward_meter_std": 0.00055393495131284, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9615384340286255, "reward_repeat_penalty_std": 0.058148376643657684, "reward_total_composite_mean": 0.9605352282524109, "reward_total_composite_std": 0.05785057693719864} {"timestamp_utc": "2026-04-12T03:29:38Z", "mode": "train", "global_step": 3174, "epoch": 0.12748523918544402, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.848484848484849e-07, "num_tokens": 7224971.0, "completions/mean_length": 36.0, "completions/min_length": 36.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.0, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9996045231819153, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9996045231819153, "rewards/total_composite/std": 0.0, "reward": 0.9996045231819153, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0017466507852077484, "sampling/sampling_logp_difference/max": 0.04659882187843323, "sampling/importance_sampling_ratio/min": 0.9647499918937683, "sampling/importance_sampling_ratio/mean": 1.0015071630477905, "sampling/importance_sampling_ratio/max": 1.0477015972137451, "entropy": 0.016203245264478028, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9996045231819153, "reward_meter_mean": 0.9996045231819153, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9996045231819153, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:29:43Z", "mode": "train", "global_step": 3175, "epoch": 0.12752540466722898, "loss": -0.0001, "grad_norm": 0.08287062495946884, "learning_rate": 3.8181818181818187e-07, "num_tokens": 7227420.0, "completions/mean_length": 132.125, "completions/min_length": 132.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.125, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.9994284510612488, "rewards/meter/std": 6.205716090335045e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994284510612488, "rewards/total_composite/std": 6.205716090335045e-06, "reward": 0.9994284510612488, "reward_std": 6.210763785929885e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008646445348858833, "sampling/sampling_logp_difference/max": 0.9603474140167236, "sampling/importance_sampling_ratio/min": 0.3827598989009857, "sampling/importance_sampling_ratio/mean": 1.0014747381210327, "sampling/importance_sampling_ratio/max": 1.300461769104004, "entropy": 0.08139224071055651, "clip_ratio/low_mean": 0.0028337890980765224, "clip_ratio/low_min": 0.0028337890980765224, "clip_ratio/high_mean": 0.0018939394503831863, "clip_ratio/high_max": 0.0018939394503831863, "clip_ratio/region_mean": 0.004727728548459709, "reward_total_mean": 0.9994284510612488, "reward_meter_mean": 0.9994284510612488, "reward_meter_std": 6.205716090335045e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994284510612488, "reward_total_composite_std": 6.205716090335045e-06} {"timestamp_utc": "2026-04-12T03:29:51Z", "mode": "train", "global_step": 3176, "epoch": 0.12756557014901393, "loss": -0.0238, "grad_norm": 1.365169644355774, "learning_rate": 3.787878787878788e-07, "num_tokens": 7232148.0, "completions/mean_length": 351.0, "completions/min_length": 317.0, "completions/max_length": 371.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 351.0, "completions/min_terminated_length": 317.0, "completions/max_terminated_length": 371.0, "rewards/meter/mean": 0.9988387823104858, "rewards/meter/std": 0.0003177140897605568, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.04629101976752281, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9736841917037964, "rewards/repeat_penalty/std": 0.03978573530912399, "rewards/total_composite/mean": 0.9475728273391724, "rewards/total_composite/std": 0.04686007648706436, "reward": 0.9475728273391724, "reward_std": 0.04686008021235466, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03374847397208214, "sampling/sampling_logp_difference/max": 4.901726722717285, "sampling/importance_sampling_ratio/min": 0.007433735765516758, "sampling/importance_sampling_ratio/mean": 1.008555293083191, "sampling/importance_sampling_ratio/max": 1.7764185667037964, "entropy": 0.3763493411242962, "clip_ratio/low_mean": 0.008392618678044528, "clip_ratio/low_min": 0.008392618678044528, "clip_ratio/high_mean": 0.012750733643770218, "clip_ratio/high_max": 0.012750733643770218, "clip_ratio/region_mean": 0.021143352321814746, "reward_total_mean": 0.9475728273391724, "reward_meter_mean": 0.9988387823104858, "reward_meter_std": 0.0003177140897605568, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.04629101976752281, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9736841917037964, "reward_repeat_penalty_std": 0.03978573530912399, "reward_total_composite_mean": 0.9475728273391724, "reward_total_composite_std": 0.04686007648706436} {"timestamp_utc": "2026-04-12T03:29:56Z", "mode": "train", "global_step": 3177, "epoch": 0.12760573563079888, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.757575757575758e-07, "num_tokens": 7233836.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00024669314734637737, "sampling/sampling_logp_difference/max": 0.0033810948953032494, "sampling/importance_sampling_ratio/min": 0.9966246485710144, "sampling/importance_sampling_ratio/mean": 1.0002270936965942, "sampling/importance_sampling_ratio/max": 1.002587914466858, "entropy": 0.001803019520593807, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:30:00Z", "mode": "train", "global_step": 3178, "epoch": 0.12764590111258384, "loss": 0.0007, "grad_norm": 1.7416008710861206, "learning_rate": 3.7272727272727274e-07, "num_tokens": 7235742.0, "completions/mean_length": 71.25, "completions/min_length": 71.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.25, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9994281530380249, "rewards/meter/std": 4.47747042926494e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994281530380249, "rewards/total_composite/std": 4.47747042926494e-05, "reward": 0.9994281530380249, "reward_std": 4.4774700654670596e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007580549921840429, "sampling/sampling_logp_difference/max": 0.3693375587463379, "sampling/importance_sampling_ratio/min": 0.691192090511322, "sampling/importance_sampling_ratio/mean": 1.004219651222229, "sampling/importance_sampling_ratio/max": 1.2836189270019531, "entropy": 0.07951132394373417, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0034966744715347886, "clip_ratio/high_max": 0.0034966744715347886, "clip_ratio/region_mean": 0.0034966744715347886, "reward_total_mean": 0.9994281530380249, "reward_meter_mean": 0.9994281530380249, "reward_meter_std": 4.47747042926494e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994281530380249, "reward_total_composite_std": 4.47747042926494e-05} {"timestamp_utc": "2026-04-12T03:30:08Z", "mode": "train", "global_step": 3179, "epoch": 0.1276860665943688, "loss": 0.0176, "grad_norm": 2.992551565170288, "learning_rate": 3.6969696969696973e-07, "num_tokens": 7239068.0, "completions/mean_length": 234.75, "completions/min_length": 224.0, "completions/max_length": 241.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 234.75, "completions/min_terminated_length": 224.0, "completions/max_terminated_length": 241.0, "rewards/meter/mean": 0.998660683631897, "rewards/meter/std": 0.0007363483309745789, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9807692170143127, "rewards/repeat_penalty/std": 0.03560846298933029, "rewards/total_composite/mean": 0.9794449806213379, "rewards/total_composite/std": 0.03524334356188774, "reward": 0.9794449806213379, "reward_std": 0.03524333983659744, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.050825342535972595, "sampling/sampling_logp_difference/max": 3.3192310333251953, "sampling/importance_sampling_ratio/min": 0.03618064150214195, "sampling/importance_sampling_ratio/mean": 1.0093677043914795, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37664885073900223, "clip_ratio/low_mean": 0.00682031805627048, "clip_ratio/low_min": 0.00682031805627048, "clip_ratio/high_mean": 0.03241610305849463, "clip_ratio/high_max": 0.03241610305849463, "clip_ratio/region_mean": 0.03923642111476511, "reward_total_mean": 0.9794449806213379, "reward_meter_mean": 0.998660683631897, "reward_meter_std": 0.0007363483309745789, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9807692170143127, "reward_repeat_penalty_std": 0.03560846298933029, "reward_total_composite_mean": 0.9794449806213379, "reward_total_composite_std": 0.03524334356188774} {"timestamp_utc": "2026-04-12T03:30:13Z", "mode": "train", "global_step": 3180, "epoch": 0.12772623207615375, "loss": -0.0004, "grad_norm": 0.22532886266708374, "learning_rate": 3.666666666666667e-07, "num_tokens": 7241707.0, "completions/mean_length": 131.875, "completions/min_length": 131.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.875, "completions/min_terminated_length": 131.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.9994198083877563, "rewards/meter/std": 2.31738686125027e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994198083877563, "rewards/total_composite/std": 2.31738686125027e-05, "reward": 0.9994198083877563, "reward_std": 2.3179616619017906e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010175392031669617, "sampling/sampling_logp_difference/max": 1.3818755149841309, "sampling/importance_sampling_ratio/min": 0.2511071562767029, "sampling/importance_sampling_ratio/mean": 1.0032905340194702, "sampling/importance_sampling_ratio/max": 1.3655613660812378, "entropy": 0.08243910130113363, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.002848137926775962, "clip_ratio/high_max": 0.002848137926775962, "clip_ratio/region_mean": 0.002848137926775962, "reward_total_mean": 0.9994198083877563, "reward_meter_mean": 0.9994198083877563, "reward_meter_std": 2.31738686125027e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994198083877563, "reward_total_composite_std": 2.31738686125027e-05} {"timestamp_utc": "2026-04-12T03:30:19Z", "mode": "train", "global_step": 3181, "epoch": 0.1277663975579387, "loss": -0.0034, "grad_norm": 0.8209965825080872, "learning_rate": 3.6363636363636366e-07, "num_tokens": 7244435.0, "completions/mean_length": 165.0, "completions/min_length": 162.0, "completions/max_length": 167.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 165.0, "completions/min_terminated_length": 162.0, "completions/max_terminated_length": 167.0, "rewards/meter/mean": 0.9993767738342285, "rewards/meter/std": 0.00010731098882388324, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993767738342285, "rewards/total_composite/std": 0.00010731098882388324, "reward": 0.9993767738342285, "reward_std": 0.00010731603106250986, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014372244477272034, "sampling/sampling_logp_difference/max": 1.1659832000732422, "sampling/importance_sampling_ratio/min": 0.31161612272262573, "sampling/importance_sampling_ratio/mean": 1.0039643049240112, "sampling/importance_sampling_ratio/max": 1.6672699451446533, "entropy": 0.12881212681531906, "clip_ratio/low_mean": 0.0030676000751554966, "clip_ratio/low_min": 0.0030676000751554966, "clip_ratio/high_mean": 0.004531931248493493, "clip_ratio/high_max": 0.004531931248493493, "clip_ratio/region_mean": 0.007599531323648989, "reward_total_mean": 0.9993767738342285, "reward_meter_mean": 0.9993767738342285, "reward_meter_std": 0.00010731098882388324, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993767738342285, "reward_total_composite_std": 0.00010731098882388324} {"timestamp_utc": "2026-04-12T03:30:23Z", "mode": "train", "global_step": 3182, "epoch": 0.12780656303972365, "loss": -0.0002, "grad_norm": 1.5534006357192993, "learning_rate": 3.6060606060606065e-07, "num_tokens": 7246300.0, "completions/mean_length": 68.125, "completions/min_length": 68.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.125, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9994250535964966, "rewards/meter/std": 0.00022647925652563572, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994250535964966, "rewards/total_composite/std": 0.00022647925652563572, "reward": 0.9994250535964966, "reward_std": 0.0002264756039949134, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007597931195050478, "sampling/sampling_logp_difference/max": 0.6480650901794434, "sampling/importance_sampling_ratio/min": 0.5230568647384644, "sampling/importance_sampling_ratio/mean": 1.0001282691955566, "sampling/importance_sampling_ratio/max": 1.234535813331604, "entropy": 0.05426653893664479, "clip_ratio/low_mean": 0.005434782709926367, "clip_ratio/low_min": 0.005434782709926367, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.005434782709926367, "reward_total_mean": 0.9994250535964966, "reward_meter_mean": 0.9994250535964966, "reward_meter_std": 0.00022647925652563572, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994250535964966, "reward_total_composite_std": 0.00022647925652563572} {"timestamp_utc": "2026-04-12T03:30:27Z", "mode": "train", "global_step": 3183, "epoch": 0.1278467285215086, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.5757575757575764e-07, "num_tokens": 7247884.0, "completions/mean_length": 35.0, "completions/min_length": 35.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9980231523513794, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980231523513794, "rewards/total_composite/std": 0.0, "reward": 0.9980231523513794, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00043650183943100274, "sampling/sampling_logp_difference/max": 0.008478658273816109, "sampling/importance_sampling_ratio/min": 1.000000238418579, "sampling/importance_sampling_ratio/mean": 1.0004373788833618, "sampling/importance_sampling_ratio/max": 1.0085147619247437, "entropy": 0.0031260672258213162, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9980231523513794, "reward_meter_mean": 0.9980231523513794, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9980231523513794, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:30:32Z", "mode": "train", "global_step": 3184, "epoch": 0.12788689400329356, "loss": -0.0004, "grad_norm": 0.10775468498468399, "learning_rate": 3.545454545454546e-07, "num_tokens": 7249747.0, "completions/mean_length": 66.875, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.875, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981533288955688, "rewards/meter/std": 3.6652184007834876e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981533288955688, "rewards/total_composite/std": 3.6652184007834876e-06, "reward": 0.9981533288955688, "reward_std": 3.6725155041494872e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007368480786681175, "sampling/sampling_logp_difference/max": 0.9039099216461182, "sampling/importance_sampling_ratio/min": 0.4049831032752991, "sampling/importance_sampling_ratio/mean": 1.000977635383606, "sampling/importance_sampling_ratio/max": 1.437882661819458, "entropy": 0.04700829554349184, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0018656715983524919, "reward_total_mean": 0.9981533288955688, "reward_meter_mean": 0.9981533288955688, "reward_meter_std": 3.6652184007834876e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981533288955688, "reward_total_composite_std": 3.6652184007834876e-06} {"timestamp_utc": "2026-04-12T03:30:36Z", "mode": "train", "global_step": 3185, "epoch": 0.12792705948507851, "loss": 0.0221, "grad_norm": 5.8911333084106445, "learning_rate": 3.515151515151515e-07, "num_tokens": 7251597.0, "completions/mean_length": 66.25, "completions/min_length": 65.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.25, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9895763397216797, "rewards/meter/std": 0.009416628628969193, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9895763397216797, "rewards/total_composite/std": 0.009416628628969193, "reward": 0.9895763397216797, "reward_std": 0.00941664818674326, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02673407271504402, "sampling/sampling_logp_difference/max": 1.822418212890625, "sampling/importance_sampling_ratio/min": 0.1616344153881073, "sampling/importance_sampling_ratio/mean": 1.0064246654510498, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15871062874794006, "clip_ratio/low_mean": 0.0071428571827709675, "clip_ratio/low_min": 0.0071428571827709675, "clip_ratio/high_mean": 0.013145374483428895, "clip_ratio/high_max": 0.013145374483428895, "clip_ratio/region_mean": 0.020288231666199863, "reward_total_mean": 0.9895763397216797, "reward_meter_mean": 0.9895763397216797, "reward_meter_std": 0.009416628628969193, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9895763397216797, "reward_total_composite_std": 0.009416628628969193} {"timestamp_utc": "2026-04-12T03:30:41Z", "mode": "train", "global_step": 3186, "epoch": 0.12796722496686347, "loss": 0.0207, "grad_norm": 7.721897125244141, "learning_rate": 3.4848484848484856e-07, "num_tokens": 7253219.0, "completions/mean_length": 45.75, "completions/min_length": 43.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.75, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9423834681510925, "rewards/meter/std": 0.005602226592600346, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9423834681510925, "rewards/total_composite/std": 0.005602226592600346, "reward": 0.9423834681510925, "reward_std": 0.005602239165455103, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.047884196043014526, "sampling/sampling_logp_difference/max": 1.721750020980835, "sampling/importance_sampling_ratio/min": 0.1787530481815338, "sampling/importance_sampling_ratio/mean": 1.0041768550872803, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.14682762045413256, "clip_ratio/low_mean": 0.005434782709926367, "clip_ratio/low_min": 0.005434782709926367, "clip_ratio/high_mean": 0.022629241226240993, "clip_ratio/high_max": 0.022629241226240993, "clip_ratio/region_mean": 0.02806402393616736, "reward_total_mean": 0.9423834681510925, "reward_meter_mean": 0.9423834681510925, "reward_meter_std": 0.005602226592600346, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9423834681510925, "reward_total_composite_std": 0.005602226592600346} {"timestamp_utc": "2026-04-12T03:30:46Z", "mode": "train", "global_step": 3187, "epoch": 0.12800739044864842, "loss": 0.0047, "grad_norm": 1.325953483581543, "learning_rate": 3.454545454545455e-07, "num_tokens": 7255248.0, "completions/mean_length": 78.625, "completions/min_length": 78.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.625, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9991061091423035, "rewards/meter/std": 8.766540122451261e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991061091423035, "rewards/total_composite/std": 8.766540122451261e-05, "reward": 0.9991061091423035, "reward_std": 8.767841063672677e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01794569194316864, "sampling/sampling_logp_difference/max": 0.9910240173339844, "sampling/importance_sampling_ratio/min": 0.3711963891983032, "sampling/importance_sampling_ratio/mean": 1.0035501718521118, "sampling/importance_sampling_ratio/max": 1.395779013633728, "entropy": 0.17396301217377186, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.00957481365185231, "clip_ratio/high_max": 0.00957481365185231, "clip_ratio/region_mean": 0.00957481365185231, "reward_total_mean": 0.9991061091423035, "reward_meter_mean": 0.9991061091423035, "reward_meter_std": 8.766540122451261e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9991061091423035, "reward_total_composite_std": 8.766540122451261e-05} {"timestamp_utc": "2026-04-12T03:30:51Z", "mode": "train", "global_step": 3188, "epoch": 0.12804755593043338, "loss": 0.0022, "grad_norm": 2.738718032836914, "learning_rate": 3.4242424242424243e-07, "num_tokens": 7257350.0, "completions/mean_length": 100.75, "completions/min_length": 99.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.75, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.9992691278457642, "rewards/meter/std": 0.00017112072964664549, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9742865562438965, "rewards/total_composite/std": 0.07064837217330933, "reward": 0.9742865562438965, "reward_std": 0.07064837962388992, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018476763740181923, "sampling/sampling_logp_difference/max": 1.5925981998443604, "sampling/importance_sampling_ratio/min": 0.20339646935462952, "sampling/importance_sampling_ratio/mean": 1.0054959058761597, "sampling/importance_sampling_ratio/max": 1.7092481851577759, "entropy": 0.1316568348556757, "clip_ratio/low_mean": 0.0012499999720603228, "clip_ratio/low_min": 0.0012499999720603228, "clip_ratio/high_mean": 0.014866774436086416, "clip_ratio/high_max": 0.014866774436086416, "clip_ratio/region_mean": 0.01611677440814674, "reward_total_mean": 0.9742865562438965, "reward_meter_mean": 0.9992691278457642, "reward_meter_std": 0.00017112072964664549, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9742865562438965, "reward_total_composite_std": 0.07064837217330933} {"timestamp_utc": "2026-04-12T03:30:55Z", "mode": "train", "global_step": 3189, "epoch": 0.12808772141221833, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.393939393939395e-07, "num_tokens": 7258606.0, "completions/mean_length": 34.0, "completions/min_length": 34.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9923644065856934, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9923644065856934, "rewards/total_composite/std": 0.0, "reward": 0.9923644065856934, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.003486229106783867, "sampling/sampling_logp_difference/max": 0.09297092258930206, "sampling/importance_sampling_ratio/min": 0.961505115032196, "sampling/importance_sampling_ratio/mean": 1.0026298761367798, "sampling/importance_sampling_ratio/max": 1.097429871559143, "entropy": 0.03649881505407393, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9923644065856934, "reward_meter_mean": 0.9923644065856934, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9923644065856934, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:30:59Z", "mode": "train", "global_step": 3190, "epoch": 0.12812788689400328, "loss": 0.0015, "grad_norm": 2.772718906402588, "learning_rate": 3.363636363636364e-07, "num_tokens": 7260528.0, "completions/mean_length": 68.25, "completions/min_length": 68.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.25, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9994656443595886, "rewards/meter/std": 8.461968536721542e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994656443595886, "rewards/total_composite/std": 8.461968536721542e-05, "reward": 0.9994656443595886, "reward_std": 8.462517871521413e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00850264448672533, "sampling/sampling_logp_difference/max": 0.919212818145752, "sampling/importance_sampling_ratio/min": 0.39883288741111755, "sampling/importance_sampling_ratio/mean": 1.003930926322937, "sampling/importance_sampling_ratio/max": 1.5606142282485962, "entropy": 0.052496086340397596, "clip_ratio/low_mean": 0.0018382353009656072, "clip_ratio/low_min": 0.0018382353009656072, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0018382353009656072, "reward_total_mean": 0.9994656443595886, "reward_meter_mean": 0.9994656443595886, "reward_meter_std": 8.461968536721542e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994656443595886, "reward_total_composite_std": 8.461968536721542e-05} {"timestamp_utc": "2026-04-12T03:31:04Z", "mode": "train", "global_step": 3191, "epoch": 0.12816805237578824, "loss": 0.0002, "grad_norm": 0.23446327447891235, "learning_rate": 3.3333333333333335e-07, "num_tokens": 7262336.0, "completions/mean_length": 71.0, "completions/min_length": 71.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.0, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9994317293167114, "rewards/meter/std": 1.128076564782532e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994317293167114, "rewards/total_composite/std": 1.128076564782532e-05, "reward": 0.9994317293167114, "reward_std": 1.127942687162431e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007548379711806774, "sampling/sampling_logp_difference/max": 0.2933727502822876, "sampling/importance_sampling_ratio/min": 0.7457441091537476, "sampling/importance_sampling_ratio/mean": 1.0047414302825928, "sampling/importance_sampling_ratio/max": 1.223402500152588, "entropy": 0.08830153662711382, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9994317293167114, "reward_meter_mean": 0.9994317293167114, "reward_meter_std": 1.128076564782532e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994317293167114, "reward_total_composite_std": 1.128076564782532e-05} {"timestamp_utc": "2026-04-12T03:31:08Z", "mode": "train", "global_step": 3192, "epoch": 0.1282082178575732, "loss": 0.0136, "grad_norm": 5.140721321105957, "learning_rate": 3.303030303030303e-07, "num_tokens": 7263826.0, "completions/mean_length": 34.25, "completions/min_length": 34.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.25, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9904930591583252, "rewards/meter/std": 0.02051379904150963, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9904930591583252, "rewards/total_composite/std": 0.02051379904150963, "reward": 0.9904930591583252, "reward_std": 0.02051379904150963, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006564823444932699, "sampling/sampling_logp_difference/max": 0.8152861595153809, "sampling/importance_sampling_ratio/min": 0.4425126910209656, "sampling/importance_sampling_ratio/mean": 1.0013935565948486, "sampling/importance_sampling_ratio/max": 1.33383309841156, "entropy": 0.05123408907093108, "clip_ratio/low_mean": 0.0069444444961845875, "clip_ratio/low_min": 0.0069444444961845875, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0069444444961845875, "reward_total_mean": 0.9904930591583252, "reward_meter_mean": 0.9904930591583252, "reward_meter_std": 0.02051379904150963, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9904930591583252, "reward_total_composite_std": 0.02051379904150963} {"timestamp_utc": "2026-04-12T03:31:12Z", "mode": "train", "global_step": 3193, "epoch": 0.12824838333935815, "loss": 0.0001, "grad_norm": 0.0006765525904484093, "learning_rate": 3.2727272727272733e-07, "num_tokens": 7265538.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9973390102386475, "rewards/meter/std": 2.954577098535083e-07, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973390102386475, "rewards/total_composite/std": 2.954577098535083e-07, "reward": 0.9973390102386475, "reward_std": 3.0140995477268007e-07, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0017861899686977267, "sampling/sampling_logp_difference/max": 0.09160566329956055, "sampling/importance_sampling_ratio/min": 0.9242770671844482, "sampling/importance_sampling_ratio/mean": 1.0007272958755493, "sampling/importance_sampling_ratio/max": 1.0959326028823853, "entropy": 0.013490123208612204, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0020491802133619785, "clip_ratio/high_max": 0.0020491802133619785, "clip_ratio/region_mean": 0.0020491802133619785, "reward_total_mean": 0.9973390102386475, "reward_meter_mean": 0.9973390102386475, "reward_meter_std": 2.954577098535083e-07, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9973390102386475, "reward_total_composite_std": 2.954577098535083e-07} {"timestamp_utc": "2026-04-12T03:31:18Z", "mode": "train", "global_step": 3194, "epoch": 0.1282885488211431, "loss": -0.0017, "grad_norm": 0.48764580488204956, "learning_rate": 3.2424242424242427e-07, "num_tokens": 7267985.0, "completions/mean_length": 131.875, "completions/min_length": 131.0, "completions/max_length": 132.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.875, "completions/min_terminated_length": 131.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.999407172203064, "rewards/meter/std": 5.462394256028347e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999407172203064, "rewards/total_composite/std": 5.462394256028347e-05, "reward": 0.999407172203064, "reward_std": 5.462476474349387e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0077711548656225204, "sampling/sampling_logp_difference/max": 0.6230001449584961, "sampling/importance_sampling_ratio/min": 0.5363329648971558, "sampling/importance_sampling_ratio/mean": 1.003739833831787, "sampling/importance_sampling_ratio/max": 1.5405017137527466, "entropy": 0.0831528240814805, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.006628788076341152, "clip_ratio/high_max": 0.006628788076341152, "clip_ratio/region_mean": 0.006628788076341152, "reward_total_mean": 0.999407172203064, "reward_meter_mean": 0.999407172203064, "reward_meter_std": 5.462394256028347e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.999407172203064, "reward_total_composite_std": 5.462394256028347e-05} {"timestamp_utc": "2026-04-12T03:31:22Z", "mode": "train", "global_step": 3195, "epoch": 0.12832871430292805, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.212121212121212e-07, "num_tokens": 7269473.0, "completions/mean_length": 32.0, "completions/min_length": 32.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "reward": 0.9992982149124146, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 6.992343696765602e-05, "sampling/sampling_logp_difference/max": 0.0013757887063547969, "sampling/importance_sampling_ratio/min": 0.9994438290596008, "sampling/importance_sampling_ratio/mean": 1.0000648498535156, "sampling/importance_sampling_ratio/max": 1.001376748085022, "entropy": 0.000906125656911172, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9992982149124146, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:31:31Z", "mode": "train", "global_step": 3196, "epoch": 0.128368879784713, "loss": -0.0218, "grad_norm": 1.7959932088851929, "learning_rate": 3.181818181818182e-07, "num_tokens": 7274421.0, "completions/mean_length": 397.5, "completions/min_length": 379.0, "completions/max_length": 424.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 397.5, "completions/min_terminated_length": 379.0, "completions/max_terminated_length": 424.0, "rewards/meter/mean": 0.9988911747932434, "rewards/meter/std": 0.0005708896787837148, "rewards/count_adherence/mean": 0.7884615659713745, "rewards/count_adherence/std": 0.03560846298933029, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9937499761581421, "rewards/repeat_penalty/std": 0.01767767407000065, "rewards/total_composite/mean": 0.782776951789856, "rewards/total_composite/std": 0.04057204723358154, "reward": 0.782776951789856, "reward_std": 0.04057204723358154, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05620747059583664, "sampling/sampling_logp_difference/max": 1.8958301544189453, "sampling/importance_sampling_ratio/min": 0.15019360184669495, "sampling/importance_sampling_ratio/mean": 1.0130902528762817, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5010012574493885, "clip_ratio/low_mean": 0.017221659421920776, "clip_ratio/low_min": 0.017221659421920776, "clip_ratio/high_mean": 0.007785308640450239, "clip_ratio/high_max": 0.007785308640450239, "clip_ratio/region_mean": 0.025006968062371016, "reward_total_mean": 0.782776951789856, "reward_meter_mean": 0.9988911747932434, "reward_meter_std": 0.0005708896787837148, "reward_count_adherence_mean": 0.7884615659713745, "reward_count_adherence_std": 0.03560846298933029, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9937499761581421, "reward_repeat_penalty_std": 0.01767767407000065, "reward_total_composite_mean": 0.782776951789856, "reward_total_composite_std": 0.04057204723358154} {"timestamp_utc": "2026-04-12T03:31:38Z", "mode": "train", "global_step": 3197, "epoch": 0.12840904526649796, "loss": 0.0732, "grad_norm": 3.9520342350006104, "learning_rate": 3.151515151515152e-07, "num_tokens": 7278424.0, "completions/mean_length": 277.375, "completions/min_length": 254.0, "completions/max_length": 313.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 277.375, "completions/min_terminated_length": 254.0, "completions/max_terminated_length": 313.0, "rewards/meter/mean": 0.9972818493843079, "rewards/meter/std": 0.001952509512193501, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0578637570142746, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9852941036224365, "rewards/repeat_penalty/std": 0.04159451276063919, "rewards/total_composite/mean": 0.953273355960846, "rewards/total_composite/std": 0.08584806323051453, "reward": 0.953273355960846, "reward_std": 0.08584806323051453, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06301229447126389, "sampling/sampling_logp_difference/max": 1.7296597957611084, "sampling/importance_sampling_ratio/min": 0.22898982465267181, "sampling/importance_sampling_ratio/mean": 1.01344633102417, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.570828665047884, "clip_ratio/low_mean": 0.012125813402235508, "clip_ratio/low_min": 0.012125813402235508, "clip_ratio/high_mean": 0.036240562330931425, "clip_ratio/high_max": 0.036240562330931425, "clip_ratio/region_mean": 0.04836637573316693, "reward_total_mean": 0.953273355960846, "reward_meter_mean": 0.9972818493843079, "reward_meter_std": 0.001952509512193501, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0578637570142746, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9852941036224365, "reward_repeat_penalty_std": 0.04159451276063919, "reward_total_composite_mean": 0.953273355960846, "reward_total_composite_std": 0.08584806323051453} {"timestamp_utc": "2026-04-12T03:31:42Z", "mode": "train", "global_step": 3198, "epoch": 0.12844921074828292, "loss": 0.0029, "grad_norm": 11.894042015075684, "learning_rate": 3.1212121212121213e-07, "num_tokens": 7280073.0, "completions/mean_length": 46.125, "completions/min_length": 44.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.125, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.9368916749954224, "rewards/meter/std": 0.018891815096139908, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9368916749954224, "rewards/total_composite/std": 0.018891815096139908, "reward": 0.9368916749954224, "reward_std": 0.018891818821430206, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03293498232960701, "sampling/sampling_logp_difference/max": 2.211425304412842, "sampling/importance_sampling_ratio/min": 0.10954440385103226, "sampling/importance_sampling_ratio/mean": 1.0053000450134277, "sampling/importance_sampling_ratio/max": 1.6582306623458862, "entropy": 0.14615605771541595, "clip_ratio/low_mean": 0.013334879651665688, "clip_ratio/low_min": 0.013334879651665688, "clip_ratio/high_mean": 0.019021739484742284, "clip_ratio/high_max": 0.019021739484742284, "clip_ratio/region_mean": 0.03235661913640797, "reward_total_mean": 0.9368916749954224, "reward_meter_mean": 0.9368916749954224, "reward_meter_std": 0.018891815096139908, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9368916749954224, "reward_total_composite_std": 0.018891815096139908} {"timestamp_utc": "2026-04-12T03:31:49Z", "mode": "train", "global_step": 3199, "epoch": 0.12848937623006787, "loss": 0.0104, "grad_norm": 1.7074098587036133, "learning_rate": 3.090909090909091e-07, "num_tokens": 7283464.0, "completions/mean_length": 201.875, "completions/min_length": 189.0, "completions/max_length": 207.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 201.875, "completions/min_terminated_length": 189.0, "completions/max_terminated_length": 207.0, "rewards/meter/mean": 0.8017416000366211, "rewards/meter/std": 0.2793266177177429, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.96875, "rewards/repeat_penalty/std": 0.0431290864944458, "rewards/total_composite/mean": 0.7810623049736023, "rewards/total_composite/std": 0.2836529314517975, "reward": 0.7810623049736023, "reward_std": 0.2836529314517975, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.032916199415922165, "sampling/sampling_logp_difference/max": 4.614783763885498, "sampling/importance_sampling_ratio/min": 0.009904325008392334, "sampling/importance_sampling_ratio/mean": 1.007271647453308, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31216721422970295, "clip_ratio/low_mean": 0.00802644295617938, "clip_ratio/low_min": 0.00802644295617938, "clip_ratio/high_mean": 0.010650901531334966, "clip_ratio/high_max": 0.010650901531334966, "clip_ratio/region_mean": 0.018677344487514347, "reward_total_mean": 0.7810623049736023, "reward_meter_mean": 0.8017416000366211, "reward_meter_std": 0.2793266177177429, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.96875, "reward_repeat_penalty_std": 0.0431290864944458, "reward_total_composite_mean": 0.7810623049736023, "reward_total_composite_std": 0.2836529314517975} {"timestamp_utc": "2026-04-12T03:31:54Z", "mode": "train", "global_step": 3200, "epoch": 0.12852954171185282, "loss": 0.0, "grad_norm": 0.4144695997238159, "learning_rate": 3.0606060606060606e-07, "num_tokens": 7285520.0, "completions/mean_length": 93.0, "completions/min_length": 93.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.0, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.9977416396141052, "rewards/meter/std": 3.231431037420407e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977416396141052, "rewards/total_composite/std": 3.231431037420407e-05, "reward": 0.9977416396141052, "reward_std": 3.2314139389200136e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010148909874260426, "sampling/sampling_logp_difference/max": 0.6211330890655518, "sampling/importance_sampling_ratio/min": 0.5517995953559875, "sampling/importance_sampling_ratio/mean": 1.0042681694030762, "sampling/importance_sampling_ratio/max": 1.8610354661941528, "entropy": 0.07502706721425056, "clip_ratio/low_mean": 0.0026881720405071974, "clip_ratio/low_min": 0.0026881720405071974, "clip_ratio/high_mean": 0.004032258060760796, "clip_ratio/high_max": 0.004032258060760796, "clip_ratio/region_mean": 0.0067204301012679935, "reward_total_mean": 0.9977416396141052, "reward_meter_mean": 0.9977416396141052, "reward_meter_std": 3.231431037420407e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9977416396141052, "reward_total_composite_std": 3.231431037420407e-05} {"timestamp_utc": "2026-04-12T03:33:12Z", "mode": "eval", "global_step": 3200, "epoch": 0.12852954171185282, "eval_loss": NaN, "eval_runtime": 78.3425, "eval_samples_per_second": 1.328, "eval_steps_per_second": 0.166, "eval_num_tokens": 7285520.0, "eval_completions/mean_length": 217.47115384615384, "eval_completions/min_length": 60.76923076923077, "eval_completions/max_length": 424.15384615384613, "eval_completions/clipped_ratio": 0.04807692307692308, "eval_completions/mean_terminated_length": 202.3791222205529, "eval_completions/min_terminated_length": 60.76923076923077, "eval_completions/max_terminated_length": 381.38461538461536, "eval_rewards/meter/mean": 0.7966891756424537, "eval_rewards/meter/std": 0.3122874472576838, "eval_rewards/count_adherence/mean": 0.962191457931812, "eval_rewards/count_adherence/std": 0.056447364103335604, "eval_rewards/arabic_clean/mean": 0.9711538461538461, "eval_rewards/arabic_clean/std": 0.06280488005051246, "eval_rewards/repeat_penalty/mean": 0.9550704268308786, "eval_rewards/repeat_penalty/std": 0.07092011161148548, "eval_rewards/total_composite/mean": 0.716111809015274, "eval_rewards/total_composite/std": 0.33144921350937623, "eval_reward": 0.716111809015274, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.03270428670713535, "eval_sampling/sampling_logp_difference/max": 1.224547532888559, "eval_sampling/importance_sampling_ratio/min": 0.3024700696651752, "eval_sampling/importance_sampling_ratio/mean": 1.0096190892733061, "eval_sampling/importance_sampling_ratio/max": 1.5303410291671753, "eval_entropy": 0.38261800087415254, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.716111809015274, "eval_reward_meter_mean": 0.7966891756424537, "eval_reward_meter_std": 0.3122874472576838, "eval_reward_count_adherence_mean": 0.962191457931812, "eval_reward_count_adherence_std": 0.056447364103335604, "eval_reward_arabic_clean_mean": 0.9711538461538461, "eval_reward_arabic_clean_std": 0.06280488005051246, "eval_reward_repeat_penalty_mean": 0.9550704268308786, "eval_reward_repeat_penalty_std": 0.07092011161148548, "eval_reward_total_composite_mean": 0.716111809015274, "eval_reward_total_composite_std": 0.33144921350937623} {"timestamp_utc": "2026-04-12T03:33:20Z", "mode": "train", "global_step": 3201, "epoch": 0.12856970719363778, "loss": 0.0005, "grad_norm": 0.3707689046859741, "learning_rate": 3.0303030303030305e-07, "num_tokens": 7287416.0, "completions/mean_length": 71.0, "completions/min_length": 71.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.0, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9994514584541321, "rewards/meter/std": 2.538687112974003e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994514584541321, "rewards/total_composite/std": 2.538687112974003e-05, "reward": 0.9994514584541321, "reward_std": 2.539291381253861e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009306100197136402, "sampling/sampling_logp_difference/max": 0.6520857810974121, "sampling/importance_sampling_ratio/min": 0.5209580659866333, "sampling/importance_sampling_ratio/mean": 1.0068362951278687, "sampling/importance_sampling_ratio/max": 1.66619074344635, "entropy": 0.07758000679314137, "clip_ratio/low_mean": 0.005281690042465925, "clip_ratio/low_min": 0.005281690042465925, "clip_ratio/high_mean": 0.0017605633474886417, "clip_ratio/high_max": 0.0017605633474886417, "clip_ratio/region_mean": 0.007042253389954567, "reward_total_mean": 0.9994514584541321, "reward_meter_mean": 0.9994514584541321, "reward_meter_std": 2.538687112974003e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994514584541321, "reward_total_composite_std": 2.538687112974003e-05} {"timestamp_utc": "2026-04-12T03:33:26Z", "mode": "train", "global_step": 3202, "epoch": 0.12860987267542273, "loss": -0.0017, "grad_norm": 2.368190050125122, "learning_rate": 3.0000000000000004e-07, "num_tokens": 7289899.0, "completions/mean_length": 128.375, "completions/min_length": 127.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.375, "completions/min_terminated_length": 127.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9979463815689087, "rewards/meter/std": 3.4329503250773996e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.962304949760437, "rewards/total_composite/std": 0.06598580628633499, "reward": 0.962304949760437, "reward_std": 0.0659857988357544, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017233658581972122, "sampling/sampling_logp_difference/max": 1.2134406566619873, "sampling/importance_sampling_ratio/min": 0.29717305302619934, "sampling/importance_sampling_ratio/mean": 1.0038442611694336, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.16156750172376633, "clip_ratio/low_mean": 0.003929318394511938, "clip_ratio/low_min": 0.003929318394511938, "clip_ratio/high_mean": 0.01261215889826417, "clip_ratio/high_max": 0.01261215889826417, "clip_ratio/region_mean": 0.016541477292776108, "reward_total_mean": 0.962304949760437, "reward_meter_mean": 0.9979463815689087, "reward_meter_std": 3.4329503250773996e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.962304949760437, "reward_total_composite_std": 0.06598580628633499} {"timestamp_utc": "2026-04-12T03:33:33Z", "mode": "train", "global_step": 3203, "epoch": 0.12865003815720769, "loss": 0.0342, "grad_norm": 2.0183863639831543, "learning_rate": 2.96969696969697e-07, "num_tokens": 7293996.0, "completions/mean_length": 281.125, "completions/min_length": 260.0, "completions/max_length": 294.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 281.125, "completions/min_terminated_length": 260.0, "completions/max_terminated_length": 294.0, "rewards/meter/mean": 0.9827648401260376, "rewards/meter/std": 0.0459895133972168, "rewards/count_adherence/mean": 0.921875, "rewards/count_adherence/std": 0.06469365209341049, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9612745046615601, "rewards/repeat_penalty/std": 0.045027956366539, "rewards/total_composite/mean": 0.8708112835884094, "rewards/total_composite/std": 0.08316829800605774, "reward": 0.8708112835884094, "reward_std": 0.08316829800605774, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.040366996079683304, "sampling/sampling_logp_difference/max": 2.540407419204712, "sampling/importance_sampling_ratio/min": 0.07883427292108536, "sampling/importance_sampling_ratio/mean": 1.0026259422302246, "sampling/importance_sampling_ratio/max": 1.6892542839050293, "entropy": 0.34676245227456093, "clip_ratio/low_mean": 0.006470522261224687, "clip_ratio/low_min": 0.006470522261224687, "clip_ratio/high_mean": 0.01714910357259214, "clip_ratio/high_max": 0.01714910357259214, "clip_ratio/region_mean": 0.023619625833816826, "reward_total_mean": 0.8708112835884094, "reward_meter_mean": 0.9827648401260376, "reward_meter_std": 0.0459895133972168, "reward_count_adherence_mean": 0.921875, "reward_count_adherence_std": 0.06469365209341049, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9612745046615601, "reward_repeat_penalty_std": 0.045027956366539, "reward_total_composite_mean": 0.8708112835884094, "reward_total_composite_std": 0.08316829800605774} {"timestamp_utc": "2026-04-12T03:33:38Z", "mode": "train", "global_step": 3204, "epoch": 0.12869020363899264, "loss": 0.0001, "grad_norm": 0.02218056656420231, "learning_rate": 2.9393939393939397e-07, "num_tokens": 7295796.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9973360300064087, "rewards/meter/std": 8.113272997434251e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973360300064087, "rewards/total_composite/std": 8.113272997434251e-06, "reward": 0.9973360300064087, "reward_std": 8.10424353403505e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.002050558337941766, "sampling/sampling_logp_difference/max": 0.22490382194519043, "sampling/importance_sampling_ratio/min": 0.8462150692939758, "sampling/importance_sampling_ratio/mean": 1.0012545585632324, "sampling/importance_sampling_ratio/max": 1.2522022724151611, "entropy": 0.014257052214816213, "clip_ratio/low_mean": 0.0020491802133619785, "clip_ratio/low_min": 0.0020491802133619785, "clip_ratio/high_mean": 0.0020491802133619785, "clip_ratio/high_max": 0.0020491802133619785, "clip_ratio/region_mean": 0.004098360426723957, "reward_total_mean": 0.9973360300064087, "reward_meter_mean": 0.9973360300064087, "reward_meter_std": 8.113272997434251e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9973360300064087, "reward_total_composite_std": 8.113272997434251e-06} {"timestamp_utc": "2026-04-12T03:33:42Z", "mode": "train", "global_step": 3205, "epoch": 0.1287303691207776, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.9090909090909096e-07, "num_tokens": 7297276.0, "completions/mean_length": 28.0, "completions/min_length": 28.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.0, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9943599104881287, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9943599104881287, "rewards/total_composite/std": 0.0, "reward": 0.9943599104881287, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0001377786829834804, "sampling/sampling_logp_difference/max": 0.0014927273150533438, "sampling/importance_sampling_ratio/min": 0.9986100196838379, "sampling/importance_sampling_ratio/mean": 1.0001215934753418, "sampling/importance_sampling_ratio/max": 1.0014938116073608, "entropy": 0.0011580012142076157, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9943599104881287, "reward_meter_mean": 0.9943599104881287, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9943599104881287, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:33:47Z", "mode": "train", "global_step": 3206, "epoch": 0.12877053460256255, "loss": 0.0027, "grad_norm": 2.458156108856201, "learning_rate": 2.878787878787879e-07, "num_tokens": 7299630.0, "completions/mean_length": 127.25, "completions/min_length": 125.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.25, "completions/min_terminated_length": 125.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.9723162651062012, "rewards/meter/std": 0.04477069899439812, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9546165466308594, "rewards/total_composite/std": 0.06121518090367317, "reward": 0.9546165466308594, "reward_std": 0.06121518462896347, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.033622466027736664, "sampling/sampling_logp_difference/max": 1.334040641784668, "sampling/importance_sampling_ratio/min": 0.2634107768535614, "sampling/importance_sampling_ratio/mean": 1.0102778673171997, "sampling/importance_sampling_ratio/max": 1.842721700668335, "entropy": 0.31193940714001656, "clip_ratio/low_mean": 0.005936880130320787, "clip_ratio/low_min": 0.005936880130320787, "clip_ratio/high_mean": 0.012698987615294755, "clip_ratio/high_max": 0.012698987615294755, "clip_ratio/region_mean": 0.018635867745615542, "reward_total_mean": 0.9546165466308594, "reward_meter_mean": 0.9723162651062012, "reward_meter_std": 0.04477069899439812, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.9546165466308594, "reward_total_composite_std": 0.06121518090367317} {"timestamp_utc": "2026-04-12T03:33:52Z", "mode": "train", "global_step": 3207, "epoch": 0.1288107000843475, "loss": 0.001, "grad_norm": 3.696357250213623, "learning_rate": 2.848484848484849e-07, "num_tokens": 7301645.0, "completions/mean_length": 92.875, "completions/min_length": 92.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.875, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.9976005554199219, "rewards/meter/std": 0.0004543311079032719, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9976005554199219, "rewards/total_composite/std": 0.0004543311079032719, "reward": 0.9976005554199219, "reward_std": 0.00045434184721671045, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009967589750885963, "sampling/sampling_logp_difference/max": 0.724400520324707, "sampling/importance_sampling_ratio/min": 0.4846150279045105, "sampling/importance_sampling_ratio/mean": 1.0031743049621582, "sampling/importance_sampling_ratio/max": 1.5881195068359375, "entropy": 0.07594481343403459, "clip_ratio/low_mean": 0.0013440860202535987, "clip_ratio/low_min": 0.0013440860202535987, "clip_ratio/high_mean": 0.004061477375216782, "clip_ratio/high_max": 0.004061477375216782, "clip_ratio/region_mean": 0.005405563395470381, "reward_total_mean": 0.9976005554199219, "reward_meter_mean": 0.9976005554199219, "reward_meter_std": 0.0004543311079032719, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9976005554199219, "reward_total_composite_std": 0.0004543311079032719} {"timestamp_utc": "2026-04-12T03:33:56Z", "mode": "train", "global_step": 3208, "epoch": 0.12885086556613246, "loss": -0.0039, "grad_norm": 2.1962215900421143, "learning_rate": 2.818181818181819e-07, "num_tokens": 7303658.0, "completions/mean_length": 92.625, "completions/min_length": 91.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.625, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.9975332021713257, "rewards/meter/std": 0.0004231163766235113, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9975332021713257, "rewards/total_composite/std": 0.0004231163766235113, "reward": 0.9975332021713257, "reward_std": 0.00042310450226068497, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010751576162874699, "sampling/sampling_logp_difference/max": 0.9464244842529297, "sampling/importance_sampling_ratio/min": 0.38812631368637085, "sampling/importance_sampling_ratio/mean": 1.0051023960113525, "sampling/importance_sampling_ratio/max": 1.477807641029358, "entropy": 0.08394959289580584, "clip_ratio/low_mean": 0.006838270695880055, "clip_ratio/low_min": 0.006838270695880055, "clip_ratio/high_mean": 0.0013440860202535987, "clip_ratio/high_max": 0.0013440860202535987, "clip_ratio/region_mean": 0.008182356716133654, "reward_total_mean": 0.9975332021713257, "reward_meter_mean": 0.9975332021713257, "reward_meter_std": 0.0004231163766235113, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9975332021713257, "reward_total_composite_std": 0.0004231163766235113} {"timestamp_utc": "2026-04-12T03:34:02Z", "mode": "train", "global_step": 3209, "epoch": 0.1288910310479174, "loss": 0.0075, "grad_norm": 1.3126479387283325, "learning_rate": 2.787878787878788e-07, "num_tokens": 7306277.0, "completions/mean_length": 159.375, "completions/min_length": 158.0, "completions/max_length": 161.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 159.375, "completions/min_terminated_length": 158.0, "completions/max_terminated_length": 161.0, "rewards/meter/mean": 0.9976499080657959, "rewards/meter/std": 0.0005515830125659704, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9444444179534912, "rewards/repeat_penalty/std": 0.059391383081674576, "rewards/total_composite/mean": 0.9422216415405273, "rewards/total_composite/std": 0.05919419974088669, "reward": 0.9422216415405273, "reward_std": 0.059194166213274, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020425979048013687, "sampling/sampling_logp_difference/max": 1.9696846008300781, "sampling/importance_sampling_ratio/min": 0.13950084149837494, "sampling/importance_sampling_ratio/mean": 1.0039074420928955, "sampling/importance_sampling_ratio/max": 1.4307658672332764, "entropy": 0.20060089603066444, "clip_ratio/low_mean": 0.004707215121015906, "clip_ratio/low_min": 0.004707215121015906, "clip_ratio/high_mean": 0.00788172468310222, "clip_ratio/high_max": 0.00788172468310222, "clip_ratio/region_mean": 0.012588939804118127, "reward_total_mean": 0.9422216415405273, "reward_meter_mean": 0.9976499080657959, "reward_meter_std": 0.0005515830125659704, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9444444179534912, "reward_repeat_penalty_std": 0.059391383081674576, "reward_total_composite_mean": 0.9422216415405273, "reward_total_composite_std": 0.05919419974088669} {"timestamp_utc": "2026-04-12T03:34:07Z", "mode": "train", "global_step": 3210, "epoch": 0.12893119652970236, "loss": -0.0001, "grad_norm": 0.4994388222694397, "learning_rate": 2.757575757575758e-07, "num_tokens": 7308194.0, "completions/mean_length": 71.625, "completions/min_length": 71.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.625, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9994117021560669, "rewards/meter/std": 7.800360617693514e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994117021560669, "rewards/total_composite/std": 7.800360617693514e-05, "reward": 0.9994117021560669, "reward_std": 7.79872716520913e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010017759166657925, "sampling/sampling_logp_difference/max": 0.9018526077270508, "sampling/importance_sampling_ratio/min": 0.4058171510696411, "sampling/importance_sampling_ratio/mean": 1.0026097297668457, "sampling/importance_sampling_ratio/max": 1.3537853956222534, "entropy": 0.08111674338579178, "clip_ratio/low_mean": 0.0017361111240461469, "clip_ratio/low_min": 0.0017361111240461469, "clip_ratio/high_mean": 0.0052327855955809355, "clip_ratio/high_max": 0.0052327855955809355, "clip_ratio/region_mean": 0.006968896719627082, "reward_total_mean": 0.9994117021560669, "reward_meter_mean": 0.9994117021560669, "reward_meter_std": 7.800360617693514e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994117021560669, "reward_total_composite_std": 7.800360617693514e-05} {"timestamp_utc": "2026-04-12T03:34:11Z", "mode": "train", "global_step": 3211, "epoch": 0.12897136201148732, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.7272727272727274e-07, "num_tokens": 7310146.0, "completions/mean_length": 80.0, "completions/min_length": 80.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.0, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "reward": 0.7190229296684265, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00030469015473499894, "sampling/sampling_logp_difference/max": 0.007013600319623947, "sampling/importance_sampling_ratio/min": 0.9983328580856323, "sampling/importance_sampling_ratio/mean": 1.0002856254577637, "sampling/importance_sampling_ratio/max": 1.0070383548736572, "entropy": 0.0023747361556161195, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7190229296684265, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:34:16Z", "mode": "train", "global_step": 3212, "epoch": 0.12901152749327227, "loss": -0.003, "grad_norm": 0.7813161611557007, "learning_rate": 2.6969696969696973e-07, "num_tokens": 7311944.0, "completions/mean_length": 66.75, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.75, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.998114824295044, "rewards/meter/std": 6.83319303789176e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998114824295044, "rewards/total_composite/std": 6.83319303789176e-05, "reward": 0.998114824295044, "reward_std": 6.832612416474149e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014992273412644863, "sampling/sampling_logp_difference/max": 1.2304191589355469, "sampling/importance_sampling_ratio/min": 0.29217007756233215, "sampling/importance_sampling_ratio/mean": 0.9998994469642639, "sampling/importance_sampling_ratio/max": 1.6796454191207886, "entropy": 0.052677970845252275, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/high_mean": 0.009384893695823848, "clip_ratio/high_max": 0.009384893695823848, "clip_ratio/region_mean": 0.013172772596590221, "reward_total_mean": 0.998114824295044, "reward_meter_mean": 0.998114824295044, "reward_meter_std": 6.83319303789176e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.998114824295044, "reward_total_composite_std": 6.83319303789176e-05} {"timestamp_utc": "2026-04-12T03:34:23Z", "mode": "train", "global_step": 3213, "epoch": 0.12905169297505723, "loss": 0.0308, "grad_norm": 1.7816143035888672, "learning_rate": 2.666666666666667e-07, "num_tokens": 7315948.0, "completions/mean_length": 315.5, "completions/min_length": 307.0, "completions/max_length": 338.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 315.5, "completions/min_terminated_length": 307.0, "completions/max_terminated_length": 338.0, "rewards/meter/mean": 0.9991602897644043, "rewards/meter/std": 0.00017255292914342135, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9921875, "rewards/repeat_penalty/std": 0.022097086533904076, "rewards/total_composite/mean": 0.9913522005081177, "rewards/total_composite/std": 0.021975873038172722, "reward": 0.9913522005081177, "reward_std": 0.021975882351398468, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04883609339594841, "sampling/sampling_logp_difference/max": 1.6247730255126953, "sampling/importance_sampling_ratio/min": 0.19695636630058289, "sampling/importance_sampling_ratio/mean": 1.0101171731948853, "sampling/importance_sampling_ratio/max": 1.7197778224945068, "entropy": 0.4627891331911087, "clip_ratio/low_mean": 0.002218934940174222, "clip_ratio/low_min": 0.002218934940174222, "clip_ratio/high_mean": 0.030003347201272845, "clip_ratio/high_max": 0.030003347201272845, "clip_ratio/region_mean": 0.03222228214144707, "reward_total_mean": 0.9913522005081177, "reward_meter_mean": 0.9991602897644043, "reward_meter_std": 0.00017255292914342135, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9921875, "reward_repeat_penalty_std": 0.022097086533904076, "reward_total_composite_mean": 0.9913522005081177, "reward_total_composite_std": 0.021975873038172722} {"timestamp_utc": "2026-04-12T03:34:29Z", "mode": "train", "global_step": 3214, "epoch": 0.12909185845684218, "loss": 0.0001, "grad_norm": 0.841699481010437, "learning_rate": 2.6363636363636366e-07, "num_tokens": 7318839.0, "completions/mean_length": 178.375, "completions/min_length": 176.0, "completions/max_length": 181.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 178.375, "completions/min_terminated_length": 176.0, "completions/max_terminated_length": 181.0, "rewards/meter/mean": 0.9990983009338379, "rewards/meter/std": 0.0001289558131247759, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990983009338379, "rewards/total_composite/std": 0.0001289558131247759, "reward": 0.9990983009338379, "reward_std": 0.00012895492545794696, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018457449972629547, "sampling/sampling_logp_difference/max": 1.2622642517089844, "sampling/importance_sampling_ratio/min": 0.2830124795436859, "sampling/importance_sampling_ratio/mean": 1.0072314739227295, "sampling/importance_sampling_ratio/max": 1.48737633228302, "entropy": 0.2543965280056, "clip_ratio/low_mean": 0.003487827139906585, "clip_ratio/low_min": 0.003487827139906585, "clip_ratio/high_mean": 0.009855842916294932, "clip_ratio/high_max": 0.009855842916294932, "clip_ratio/region_mean": 0.013343670056201518, "reward_total_mean": 0.9990983009338379, "reward_meter_mean": 0.9990983009338379, "reward_meter_std": 0.0001289558131247759, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990983009338379, "reward_total_composite_std": 0.0001289558131247759} {"timestamp_utc": "2026-04-12T03:34:34Z", "mode": "train", "global_step": 3215, "epoch": 0.12913202393862713, "loss": 0.0005, "grad_norm": 0.2847835123538971, "learning_rate": 2.6060606060606065e-07, "num_tokens": 7320585.0, "completions/mean_length": 71.25, "completions/min_length": 71.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.25, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9994255304336548, "rewards/meter/std": 1.9644281564978883e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994255304336548, "rewards/total_composite/std": 1.9644281564978883e-05, "reward": 0.9994255304336548, "reward_std": 1.9626990251708776e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012419765815138817, "sampling/sampling_logp_difference/max": 1.4161691665649414, "sampling/importance_sampling_ratio/min": 0.24264176189899445, "sampling/importance_sampling_ratio/mean": 1.001387357711792, "sampling/importance_sampling_ratio/max": 1.3071258068084717, "entropy": 0.08346446044743061, "clip_ratio/low_mean": 0.0035211266949772835, "clip_ratio/low_min": 0.0035211266949772835, "clip_ratio/high_mean": 0.007017801166512072, "clip_ratio/high_max": 0.007017801166512072, "clip_ratio/region_mean": 0.010538927861489356, "reward_total_mean": 0.9994255304336548, "reward_meter_mean": 0.9994255304336548, "reward_meter_std": 1.9644281564978883e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994255304336548, "reward_total_composite_std": 1.9644281564978883e-05} {"timestamp_utc": "2026-04-12T03:34:38Z", "mode": "train", "global_step": 3216, "epoch": 0.1291721894204121, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.575757575757576e-07, "num_tokens": 7321969.0, "completions/mean_length": 29.0, "completions/min_length": 29.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9957436919212341, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957436919212341, "rewards/total_composite/std": 0.0, "reward": 0.9957436919212341, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00043711578473448753, "sampling/sampling_logp_difference/max": 0.022034302353858948, "sampling/importance_sampling_ratio/min": 0.9782066345214844, "sampling/importance_sampling_ratio/mean": 1.0002456903457642, "sampling/importance_sampling_ratio/max": 1.0145117044448853, "entropy": 0.003761589468922466, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9957436919212341, "reward_meter_mean": 0.9957436919212341, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9957436919212341, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:34:43Z", "mode": "train", "global_step": 3217, "epoch": 0.12921235490219704, "loss": 0.0004, "grad_norm": 0.06737548857927322, "learning_rate": 2.545454545454546e-07, "num_tokens": 7324185.0, "completions/mean_length": 98.0, "completions/min_length": 98.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.0, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9994082450866699, "rewards/meter/std": 3.3991077543760184e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994082450866699, "rewards/total_composite/std": 3.3991077543760184e-06, "reward": 0.9994082450866699, "reward_std": 3.404888502700487e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003399990499019623, "sampling/sampling_logp_difference/max": 0.3715987205505371, "sampling/importance_sampling_ratio/min": 0.689630925655365, "sampling/importance_sampling_ratio/mean": 1.0018640756607056, "sampling/importance_sampling_ratio/max": 1.321240782737732, "entropy": 0.035912956576794386, "clip_ratio/low_mean": 0.0012755101779475808, "clip_ratio/low_min": 0.0012755101779475808, "clip_ratio/high_mean": 0.0012755101779475808, "clip_ratio/high_max": 0.0012755101779475808, "clip_ratio/region_mean": 0.0025510203558951616, "reward_total_mean": 0.9994082450866699, "reward_meter_mean": 0.9994082450866699, "reward_meter_std": 3.3991077543760184e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994082450866699, "reward_total_composite_std": 3.3991077543760184e-06} {"timestamp_utc": "2026-04-12T03:34:48Z", "mode": "train", "global_step": 3218, "epoch": 0.129252520383982, "loss": -0.0009, "grad_norm": 1.3049845695495605, "learning_rate": 2.515151515151515e-07, "num_tokens": 7326588.0, "completions/mean_length": 141.375, "completions/min_length": 139.0, "completions/max_length": 143.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 141.375, "completions/min_terminated_length": 139.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.9991159439086914, "rewards/meter/std": 0.00015829414769541472, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991159439086914, "rewards/total_composite/std": 0.00015829414769541472, "reward": 0.9991159439086914, "reward_std": 0.00015829637413844466, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020292429253458977, "sampling/sampling_logp_difference/max": 0.8954935073852539, "sampling/importance_sampling_ratio/min": 0.40840601921081543, "sampling/importance_sampling_ratio/mean": 1.0080312490463257, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.21374017372727394, "clip_ratio/low_mean": 0.009702177019789815, "clip_ratio/low_min": 0.009702177019789815, "clip_ratio/high_mean": 0.0061869441997259855, "clip_ratio/high_max": 0.0061869441997259855, "clip_ratio/region_mean": 0.0158891212195158, "reward_total_mean": 0.9991159439086914, "reward_meter_mean": 0.9991159439086914, "reward_meter_std": 0.00015829414769541472, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9991159439086914, "reward_total_composite_std": 0.00015829414769541472} {"timestamp_utc": "2026-04-12T03:34:53Z", "mode": "train", "global_step": 3219, "epoch": 0.12929268586576695, "loss": 0.0279, "grad_norm": 5.071619510650635, "learning_rate": 2.484848484848485e-07, "num_tokens": 7328757.0, "completions/mean_length": 105.125, "completions/min_length": 102.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.125, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.9915302991867065, "rewards/meter/std": 0.006453692447394133, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9915302991867065, "rewards/total_composite/std": 0.006453692447394133, "reward": 0.9915302991867065, "reward_std": 0.006453694310039282, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0352078415453434, "sampling/sampling_logp_difference/max": 0.7839181423187256, "sampling/importance_sampling_ratio/min": 0.4566134214401245, "sampling/importance_sampling_ratio/mean": 1.0104427337646484, "sampling/importance_sampling_ratio/max": 1.9355528354644775, "entropy": 0.3442672826349735, "clip_ratio/low_mean": 0.0102413734421134, "clip_ratio/low_min": 0.0102413734421134, "clip_ratio/high_mean": 0.017989110434427857, "clip_ratio/high_max": 0.017989110434427857, "clip_ratio/region_mean": 0.028230483876541257, "reward_total_mean": 0.9915302991867065, "reward_meter_mean": 0.9915302991867065, "reward_meter_std": 0.006453692447394133, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9915302991867065, "reward_total_composite_std": 0.006453692447394133} {"timestamp_utc": "2026-04-12T03:34:58Z", "mode": "train", "global_step": 3220, "epoch": 0.1293328513475519, "loss": 0.0085, "grad_norm": 3.3606433868408203, "learning_rate": 2.4545454545454545e-07, "num_tokens": 7330682.0, "completions/mean_length": 68.625, "completions/min_length": 65.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.625, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9950666427612305, "rewards/meter/std": 0.0027368878945708275, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9950666427612305, "rewards/total_composite/std": 0.0027368878945708275, "reward": 0.9950666427612305, "reward_std": 0.0027369032613933086, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024690676480531693, "sampling/sampling_logp_difference/max": 0.7779765129089355, "sampling/importance_sampling_ratio/min": 0.45933452248573303, "sampling/importance_sampling_ratio/mean": 1.0102657079696655, "sampling/importance_sampling_ratio/max": 1.609100103378296, "entropy": 0.21459667198359966, "clip_ratio/low_mean": 0.005359299597330391, "clip_ratio/low_min": 0.005359299597330391, "clip_ratio/high_mean": 0.014640397042967379, "clip_ratio/high_max": 0.014640397042967379, "clip_ratio/region_mean": 0.01999969664029777, "reward_total_mean": 0.9950666427612305, "reward_meter_mean": 0.9950666427612305, "reward_meter_std": 0.0027368878945708275, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9950666427612305, "reward_total_composite_std": 0.0027368878945708275} {"timestamp_utc": "2026-04-12T03:35:02Z", "mode": "train", "global_step": 3221, "epoch": 0.12937301682933686, "loss": 0.0021, "grad_norm": 3.8582441806793213, "learning_rate": 2.4242424242424244e-07, "num_tokens": 7332402.0, "completions/mean_length": 71.0, "completions/min_length": 69.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.0, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9949785470962524, "rewards/meter/std": 0.0019115530885756016, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9949785470962524, "rewards/total_composite/std": 0.0019115530885756016, "reward": 0.9949785470962524, "reward_std": 0.0019115599570795894, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03391466289758682, "sampling/sampling_logp_difference/max": 1.395737648010254, "sampling/importance_sampling_ratio/min": 0.24765028059482574, "sampling/importance_sampling_ratio/mean": 1.0079292058944702, "sampling/importance_sampling_ratio/max": 1.7042900323867798, "entropy": 0.2918657101690769, "clip_ratio/low_mean": 0.024722653324715793, "clip_ratio/low_min": 0.024722653324715793, "clip_ratio/high_mean": 0.012232337845489383, "clip_ratio/high_max": 0.012232337845489383, "clip_ratio/region_mean": 0.036954991170205176, "reward_total_mean": 0.9949785470962524, "reward_meter_mean": 0.9949785470962524, "reward_meter_std": 0.0019115530885756016, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9949785470962524, "reward_total_composite_std": 0.0019115530885756016} {"timestamp_utc": "2026-04-12T03:35:08Z", "mode": "train", "global_step": 3222, "epoch": 0.1294131823111218, "loss": 0.0076, "grad_norm": 2.1812803745269775, "learning_rate": 2.3939393939393943e-07, "num_tokens": 7335408.0, "completions/mean_length": 197.75, "completions/min_length": 194.0, "completions/max_length": 205.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 197.75, "completions/min_terminated_length": 194.0, "completions/max_terminated_length": 205.0, "rewards/meter/mean": 0.9989155530929565, "rewards/meter/std": 0.00026041135424748063, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989155530929565, "rewards/total_composite/std": 0.00026041135424748063, "reward": 0.9989155530929565, "reward_std": 0.0002604102483019233, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04675072431564331, "sampling/sampling_logp_difference/max": 1.7516670227050781, "sampling/importance_sampling_ratio/min": 0.17348448932170868, "sampling/importance_sampling_ratio/mean": 1.0105738639831543, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36750227957963943, "clip_ratio/low_mean": 0.018319292226806283, "clip_ratio/low_min": 0.018319292226806283, "clip_ratio/high_mean": 0.01708377245813608, "clip_ratio/high_max": 0.01708377245813608, "clip_ratio/region_mean": 0.035403064684942365, "reward_total_mean": 0.9989155530929565, "reward_meter_mean": 0.9989155530929565, "reward_meter_std": 0.00026041135424748063, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9989155530929565, "reward_total_composite_std": 0.00026041135424748063} {"timestamp_utc": "2026-04-12T03:35:13Z", "mode": "train", "global_step": 3223, "epoch": 0.12945334779290676, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.3636363636363637e-07, "num_tokens": 7336960.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00022982771042734385, "sampling/sampling_logp_difference/max": 0.006922472268342972, "sampling/importance_sampling_ratio/min": 0.9931014776229858, "sampling/importance_sampling_ratio/mean": 1.0001906156539917, "sampling/importance_sampling_ratio/max": 1.0026967525482178, "entropy": 0.0017184649041155353, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:35:17Z", "mode": "train", "global_step": 3224, "epoch": 0.12949351327469172, "loss": -0.0014, "grad_norm": 2.2346813678741455, "learning_rate": 2.3333333333333336e-07, "num_tokens": 7339001.0, "completions/mean_length": 78.125, "completions/min_length": 77.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.125, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9987478256225586, "rewards/meter/std": 0.000937741482630372, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9987478256225586, "rewards/total_composite/std": 0.000937741482630372, "reward": 0.9987478256225586, "reward_std": 0.0009377306560054421, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016266027465462685, "sampling/sampling_logp_difference/max": 0.7468376159667969, "sampling/importance_sampling_ratio/min": 0.4738627076148987, "sampling/importance_sampling_ratio/mean": 1.0025646686553955, "sampling/importance_sampling_ratio/max": 1.5145747661590576, "entropy": 0.1890299841761589, "clip_ratio/low_mean": 0.004746835213154554, "clip_ratio/low_min": 0.004746835213154554, "clip_ratio/high_mean": 0.006410256493836641, "clip_ratio/high_max": 0.006410256493836641, "clip_ratio/region_mean": 0.011157091706991196, "reward_total_mean": 0.9987478256225586, "reward_meter_mean": 0.9987478256225586, "reward_meter_std": 0.000937741482630372, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9987478256225586, "reward_total_composite_std": 0.000937741482630372} {"timestamp_utc": "2026-04-12T03:35:22Z", "mode": "train", "global_step": 3225, "epoch": 0.12953367875647667, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.3030303030303032e-07, "num_tokens": 7340889.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 6.631435098825023e-05, "sampling/sampling_logp_difference/max": 0.005175924859941006, "sampling/importance_sampling_ratio/min": 0.9948374629020691, "sampling/importance_sampling_ratio/mean": 1.0000407695770264, "sampling/importance_sampling_ratio/max": 1.0014703273773193, "entropy": 0.0005534949777938891, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:35:26Z", "mode": "train", "global_step": 3226, "epoch": 0.12957384423826163, "loss": 0.0043, "grad_norm": 5.847517967224121, "learning_rate": 2.2727272727272729e-07, "num_tokens": 7342490.0, "completions/mean_length": 46.125, "completions/min_length": 46.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.125, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9431781768798828, "rewards/meter/std": 0.0021464419551193714, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9431781768798828, "rewards/total_composite/std": 0.0021464419551193714, "reward": 0.9431781768798828, "reward_std": 0.002146440092474222, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020598582923412323, "sampling/sampling_logp_difference/max": 1.1673927307128906, "sampling/importance_sampling_ratio/min": 0.31117722392082214, "sampling/importance_sampling_ratio/mean": 1.005907416343689, "sampling/importance_sampling_ratio/max": 1.9484820365905762, "entropy": 0.09778465609997511, "clip_ratio/low_mean": 0.019021739484742284, "clip_ratio/low_min": 0.019021739484742284, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.019021739484742284, "reward_total_mean": 0.9431781768798828, "reward_meter_mean": 0.9431781768798828, "reward_meter_std": 0.0021464419551193714, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9431781768798828, "reward_total_composite_std": 0.0021464419551193714} {"timestamp_utc": "2026-04-12T03:35:31Z", "mode": "train", "global_step": 3227, "epoch": 0.12961400972004658, "loss": -0.0002, "grad_norm": 0.25633835792541504, "learning_rate": 2.2424242424242425e-07, "num_tokens": 7344170.0, "completions/mean_length": 67.0, "completions/min_length": 67.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981385469436646, "rewards/meter/std": 1.2708615031442605e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981385469436646, "rewards/total_composite/std": 1.2708615031442605e-05, "reward": 0.9981385469436646, "reward_std": 1.2711200724879745e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007955299690365791, "sampling/sampling_logp_difference/max": 0.6916778087615967, "sampling/importance_sampling_ratio/min": 0.6211740374565125, "sampling/importance_sampling_ratio/mean": 1.002211332321167, "sampling/importance_sampling_ratio/max": 1.997063398361206, "entropy": 0.047321764286607504, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/high_mean": 0.0037313431967049837, "clip_ratio/high_max": 0.0037313431967049837, "clip_ratio/region_mean": 0.005597014795057476, "reward_total_mean": 0.9981385469436646, "reward_meter_mean": 0.9981385469436646, "reward_meter_std": 1.2708615031442605e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981385469436646, "reward_total_composite_std": 1.2708615031442605e-05} {"timestamp_utc": "2026-04-12T03:35:35Z", "mode": "train", "global_step": 3228, "epoch": 0.12965417520183153, "loss": 0.0145, "grad_norm": 8.390018463134766, "learning_rate": 2.2121212121212124e-07, "num_tokens": 7345827.0, "completions/mean_length": 46.125, "completions/min_length": 46.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.125, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9406979084014893, "rewards/meter/std": 0.006258478853851557, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9406979084014893, "rewards/total_composite/std": 0.006258478853851557, "reward": 0.9406979084014893, "reward_std": 0.0062584723345935345, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02605927735567093, "sampling/sampling_logp_difference/max": 2.4645028114318848, "sampling/importance_sampling_ratio/min": 0.08505111932754517, "sampling/importance_sampling_ratio/mean": 0.996314287185669, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07639594282954931, "clip_ratio/low_mean": 0.002659574383869767, "clip_ratio/low_min": 0.002659574383869767, "clip_ratio/high_mean": 0.00815217406488955, "clip_ratio/high_max": 0.00815217406488955, "clip_ratio/region_mean": 0.010811748448759317, "reward_total_mean": 0.9406979084014893, "reward_meter_mean": 0.9406979084014893, "reward_meter_std": 0.006258478853851557, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9406979084014893, "reward_total_composite_std": 0.006258478853851557} {"timestamp_utc": "2026-04-12T03:35:40Z", "mode": "train", "global_step": 3229, "epoch": 0.1296943406836165, "loss": -0.0021, "grad_norm": 0.8496602773666382, "learning_rate": 2.181818181818182e-07, "num_tokens": 7348082.0, "completions/mean_length": 97.875, "completions/min_length": 97.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.875, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9979760646820068, "rewards/meter/std": 7.630670006619766e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979760646820068, "rewards/total_composite/std": 7.630670006619766e-05, "reward": 0.9979760646820068, "reward_std": 7.630588515894488e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01204567588865757, "sampling/sampling_logp_difference/max": 0.7632923126220703, "sampling/importance_sampling_ratio/min": 0.46612924337387085, "sampling/importance_sampling_ratio/mean": 1.0022556781768799, "sampling/importance_sampling_ratio/max": 1.3757518529891968, "entropy": 0.10427160188555717, "clip_ratio/low_mean": 0.008941720938310027, "clip_ratio/low_min": 0.008941720938310027, "clip_ratio/high_mean": 0.005102040828205645, "clip_ratio/high_max": 0.005102040828205645, "clip_ratio/region_mean": 0.014043761766515672, "reward_total_mean": 0.9979760646820068, "reward_meter_mean": 0.9979760646820068, "reward_meter_std": 7.630670006619766e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9979760646820068, "reward_total_composite_std": 7.630670006619766e-05} {"timestamp_utc": "2026-04-12T03:35:44Z", "mode": "train", "global_step": 3230, "epoch": 0.12973450616540144, "loss": -0.0006, "grad_norm": 0.018652301281690598, "learning_rate": 2.1515151515151517e-07, "num_tokens": 7349856.0, "completions/mean_length": 66.75, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.75, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981513023376465, "rewards/meter/std": 1.5048592558741802e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981513023376465, "rewards/total_composite/std": 1.5048592558741802e-06, "reward": 0.9981513023376465, "reward_std": 1.5035095657367492e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006907826755195856, "sampling/sampling_logp_difference/max": 0.5330893993377686, "sampling/importance_sampling_ratio/min": 0.5867893099784851, "sampling/importance_sampling_ratio/mean": 1.0017226934432983, "sampling/importance_sampling_ratio/max": 1.4481335878372192, "entropy": 0.04502238845452666, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/high_mean": 0.003759611048735678, "clip_ratio/high_max": 0.003759611048735678, "clip_ratio/region_mean": 0.00562528264708817, "reward_total_mean": 0.9981513023376465, "reward_meter_mean": 0.9981513023376465, "reward_meter_std": 1.5048592558741802e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981513023376465, "reward_total_composite_std": 1.5048592558741802e-06} {"timestamp_utc": "2026-04-12T03:35:49Z", "mode": "train", "global_step": 3231, "epoch": 0.1297746716471864, "loss": -0.0038, "grad_norm": 1.449289321899414, "learning_rate": 2.1212121212121216e-07, "num_tokens": 7352000.0, "completions/mean_length": 107.0, "completions/min_length": 106.0, "completions/max_length": 110.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.0, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 110.0, "rewards/meter/mean": 0.9992383718490601, "rewards/meter/std": 0.00011975476081715897, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992383718490601, "rewards/total_composite/std": 0.00011975476081715897, "reward": 0.9992383718490601, "reward_std": 0.0001197509845951572, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015148649923503399, "sampling/sampling_logp_difference/max": 0.9686565399169922, "sampling/importance_sampling_ratio/min": 0.3795926868915558, "sampling/importance_sampling_ratio/mean": 1.0047478675842285, "sampling/importance_sampling_ratio/max": 1.8299890756607056, "entropy": 0.15556516498327255, "clip_ratio/low_mean": 0.004716981202363968, "clip_ratio/low_min": 0.004716981202363968, "clip_ratio/high_mean": 0.006999526987783611, "clip_ratio/high_max": 0.006999526987783611, "clip_ratio/region_mean": 0.011716508190147579, "reward_total_mean": 0.9992383718490601, "reward_meter_mean": 0.9992383718490601, "reward_meter_std": 0.00011975476081715897, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992383718490601, "reward_total_composite_std": 0.00011975476081715897} {"timestamp_utc": "2026-04-12T03:35:54Z", "mode": "train", "global_step": 3232, "epoch": 0.12981483712897135, "loss": -0.0004, "grad_norm": 2.6132781505584717, "learning_rate": 2.090909090909091e-07, "num_tokens": 7354036.0, "completions/mean_length": 77.5, "completions/min_length": 75.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.5, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9991195797920227, "rewards/meter/std": 0.0001867539540398866, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991195797920227, "rewards/total_composite/std": 0.0001867539540398866, "reward": 0.9991195797920227, "reward_std": 0.00018676480976864696, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022667229175567627, "sampling/sampling_logp_difference/max": 0.9247050285339355, "sampling/importance_sampling_ratio/min": 0.39664843678474426, "sampling/importance_sampling_ratio/mean": 1.005802035331726, "sampling/importance_sampling_ratio/max": 1.5898345708847046, "entropy": 0.19562415033578873, "clip_ratio/low_mean": 0.0032681134762242436, "clip_ratio/low_min": 0.0032681134762242436, "clip_ratio/high_mean": 0.009743589907884598, "clip_ratio/high_max": 0.009743589907884598, "clip_ratio/region_mean": 0.013011703384108841, "reward_total_mean": 0.9991195797920227, "reward_meter_mean": 0.9991195797920227, "reward_meter_std": 0.0001867539540398866, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9991195797920227, "reward_total_composite_std": 0.0001867539540398866} {"timestamp_utc": "2026-04-12T03:35:59Z", "mode": "train", "global_step": 3233, "epoch": 0.1298550026107563, "loss": -0.0061, "grad_norm": 3.281510591506958, "learning_rate": 2.060606060606061e-07, "num_tokens": 7356649.0, "completions/mean_length": 128.625, "completions/min_length": 127.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.625, "completions/min_terminated_length": 127.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9830908179283142, "rewards/meter/std": 0.04200791195034981, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9830908179283142, "rewards/total_composite/std": 0.04200791195034981, "reward": 0.9830908179283142, "reward_std": 0.042007915675640106, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016228584572672844, "sampling/sampling_logp_difference/max": 0.8928537368774414, "sampling/importance_sampling_ratio/min": 0.4094855487346649, "sampling/importance_sampling_ratio/mean": 1.00603187084198, "sampling/importance_sampling_ratio/max": 1.564560890197754, "entropy": 0.15214181691408157, "clip_ratio/low_mean": 0.0009842519648373127, "clip_ratio/low_min": 0.0009842519648373127, "clip_ratio/high_mean": 0.00582152372226119, "clip_ratio/high_max": 0.00582152372226119, "clip_ratio/region_mean": 0.006805775687098503, "reward_total_mean": 0.9830908179283142, "reward_meter_mean": 0.9830908179283142, "reward_meter_std": 0.04200791195034981, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9830908179283142, "reward_total_composite_std": 0.04200791195034981} {"timestamp_utc": "2026-04-12T03:36:03Z", "mode": "train", "global_step": 3234, "epoch": 0.12989516809254126, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.0303030303030303e-07, "num_tokens": 7358433.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002218023146269843, "sampling/sampling_logp_difference/max": 0.0019890288822352886, "sampling/importance_sampling_ratio/min": 0.9995585680007935, "sampling/importance_sampling_ratio/mean": 1.0002152919769287, "sampling/importance_sampling_ratio/max": 1.0019910335540771, "entropy": 0.001778529680450447, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:36:09Z", "mode": "train", "global_step": 3235, "epoch": 0.1299353335743262, "loss": 0.0024, "grad_norm": 2.066392660140991, "learning_rate": 2.0000000000000002e-07, "num_tokens": 7361110.0, "completions/mean_length": 133.625, "completions/min_length": 130.0, "completions/max_length": 137.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 133.625, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.9991873502731323, "rewards/meter/std": 0.00011400806397432461, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9813442230224609, "rewards/total_composite/std": 0.05045589804649353, "reward": 0.9813442230224609, "reward_std": 0.05045590177178383, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031857188791036606, "sampling/sampling_logp_difference/max": 2.231940269470215, "sampling/importance_sampling_ratio/min": 0.10731999576091766, "sampling/importance_sampling_ratio/mean": 1.0075880289077759, "sampling/importance_sampling_ratio/max": 1.8417998552322388, "entropy": 0.2376344669610262, "clip_ratio/low_mean": 0.001879699295386672, "clip_ratio/low_min": 0.001879699295386672, "clip_ratio/high_mean": 0.025164754129946232, "clip_ratio/high_max": 0.025164754129946232, "clip_ratio/region_mean": 0.027044453425332904, "reward_total_mean": 0.9813442230224609, "reward_meter_mean": 0.9991873502731323, "reward_meter_std": 0.00011400806397432461, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.9813442230224609, "reward_total_composite_std": 0.05045589804649353} {"timestamp_utc": "2026-04-12T03:36:16Z", "mode": "train", "global_step": 3236, "epoch": 0.12997549905611117, "loss": -0.0018, "grad_norm": 2.294574022293091, "learning_rate": 1.9696969696969698e-07, "num_tokens": 7365186.0, "completions/mean_length": 292.5, "completions/min_length": 278.0, "completions/max_length": 298.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 292.5, "completions/min_terminated_length": 278.0, "completions/max_terminated_length": 298.0, "rewards/meter/mean": 0.9674590826034546, "rewards/meter/std": 0.049329616129398346, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.9440359473228455, "rewards/repeat_penalty/std": 0.042011942714452744, "rewards/total_composite/mean": 0.6899652481079102, "rewards/total_composite/std": 0.280244380235672, "reward": 0.6899652481079102, "reward_std": 0.280244380235672, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03874696418642998, "sampling/sampling_logp_difference/max": 2.491647720336914, "sampling/importance_sampling_ratio/min": 0.08277346938848495, "sampling/importance_sampling_ratio/mean": 1.0074106454849243, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3465934246778488, "clip_ratio/low_mean": 0.002595155732706189, "clip_ratio/low_min": 0.002595155732706189, "clip_ratio/high_mean": 0.02566183707676828, "clip_ratio/high_max": 0.02566183707676828, "clip_ratio/region_mean": 0.028256992809474468, "reward_total_mean": 0.6899652481079102, "reward_meter_mean": 0.9674590826034546, "reward_meter_std": 0.049329616129398346, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.9440359473228455, "reward_repeat_penalty_std": 0.042011942714452744, "reward_total_composite_mean": 0.6899652481079102, "reward_total_composite_std": 0.280244380235672} {"timestamp_utc": "2026-04-12T03:36:20Z", "mode": "train", "global_step": 3237, "epoch": 0.13001566453789612, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.9393939393939395e-07, "num_tokens": 7366962.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 5.0931172154378146e-05, "sampling/sampling_logp_difference/max": 0.0009334206115454435, "sampling/importance_sampling_ratio/min": 0.9994771480560303, "sampling/importance_sampling_ratio/mean": 1.0000455379486084, "sampling/importance_sampling_ratio/max": 1.0009338855743408, "entropy": 0.0005204225053603295, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:36:26Z", "mode": "train", "global_step": 3238, "epoch": 0.13005583001968107, "loss": -0.0084, "grad_norm": 4.2172322273254395, "learning_rate": 1.9090909090909094e-07, "num_tokens": 7369424.0, "completions/mean_length": 133.75, "completions/min_length": 129.0, "completions/max_length": 138.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 133.75, "completions/min_terminated_length": 129.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.9989288449287415, "rewards/meter/std": 0.0006227882695384324, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989288449287415, "rewards/total_composite/std": 0.0006227882695384324, "reward": 0.9989288449287415, "reward_std": 0.0006227805861271918, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.035977549850940704, "sampling/sampling_logp_difference/max": 5.998313903808594, "sampling/importance_sampling_ratio/min": 0.0024829350877553225, "sampling/importance_sampling_ratio/mean": 1.0042328834533691, "sampling/importance_sampling_ratio/max": 1.8821479082107544, "entropy": 0.22379975207149982, "clip_ratio/low_mean": 0.007562352577224374, "clip_ratio/low_min": 0.007562352577224374, "clip_ratio/high_mean": 0.011160198540892452, "clip_ratio/high_max": 0.011160198540892452, "clip_ratio/region_mean": 0.018722551118116826, "reward_total_mean": 0.9989288449287415, "reward_meter_mean": 0.9989288449287415, "reward_meter_std": 0.0006227882695384324, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9989288449287415, "reward_total_composite_std": 0.0006227882695384324} {"timestamp_utc": "2026-04-12T03:36:30Z", "mode": "train", "global_step": 3239, "epoch": 0.13009599550146603, "loss": -0.0012, "grad_norm": 1.0152816772460938, "learning_rate": 1.878787878787879e-07, "num_tokens": 7371177.0, "completions/mean_length": 68.125, "completions/min_length": 68.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.125, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.999480128288269, "rewards/meter/std": 7.362089672824368e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999480128288269, "rewards/total_composite/std": 7.362089672824368e-05, "reward": 0.999480128288269, "reward_std": 7.363814802374691e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011260369792580605, "sampling/sampling_logp_difference/max": 1.374150037765503, "sampling/importance_sampling_ratio/min": 0.2629310190677643, "sampling/importance_sampling_ratio/mean": 1.0008584260940552, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.04902258049696684, "clip_ratio/low_mean": 0.0018382353009656072, "clip_ratio/low_min": 0.0018382353009656072, "clip_ratio/high_mean": 0.0018382353009656072, "clip_ratio/high_max": 0.0018382353009656072, "clip_ratio/region_mean": 0.0036764706019312143, "reward_total_mean": 0.999480128288269, "reward_meter_mean": 0.999480128288269, "reward_meter_std": 7.362089672824368e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.999480128288269, "reward_total_composite_std": 7.362089672824368e-05} {"timestamp_utc": "2026-04-12T03:36:38Z", "mode": "train", "global_step": 3240, "epoch": 0.13013616098325098, "loss": -0.0318, "grad_norm": 1.8787816762924194, "learning_rate": 1.8484848484848486e-07, "num_tokens": 7375719.0, "completions/mean_length": 350.75, "completions/min_length": 333.0, "completions/max_length": 380.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 350.75, "completions/min_terminated_length": 333.0, "completions/max_terminated_length": 380.0, "rewards/meter/mean": 0.9988770484924316, "rewards/meter/std": 0.0009327766601927578, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.05345224589109421, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9926470518112183, "rewards/repeat_penalty/std": 0.020797256380319595, "rewards/total_composite/mean": 0.8923759460449219, "rewards/total_composite/std": 0.056543249636888504, "reward": 0.8923759460449219, "reward_std": 0.056543245911598206, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05698416754603386, "sampling/sampling_logp_difference/max": 4.723583698272705, "sampling/importance_sampling_ratio/min": 0.008883286267518997, "sampling/importance_sampling_ratio/mean": 1.0113011598587036, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.49749621003866196, "clip_ratio/low_mean": 0.005920994793996215, "clip_ratio/low_min": 0.005920994793996215, "clip_ratio/high_mean": 0.024295849725604057, "clip_ratio/high_max": 0.024295849725604057, "clip_ratio/region_mean": 0.030216844519600272, "reward_total_mean": 0.8923759460449219, "reward_meter_mean": 0.9988770484924316, "reward_meter_std": 0.0009327766601927578, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.05345224589109421, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9926470518112183, "reward_repeat_penalty_std": 0.020797256380319595, "reward_total_composite_mean": 0.8923759460449219, "reward_total_composite_std": 0.056543249636888504} {"timestamp_utc": "2026-04-12T03:36:47Z", "mode": "train", "global_step": 3241, "epoch": 0.13017632646503594, "loss": -0.0301, "grad_norm": 1.581771969795227, "learning_rate": 1.8181818181818183e-07, "num_tokens": 7380120.0, "completions/mean_length": 353.125, "completions/min_length": 342.0, "completions/max_length": 392.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 353.125, "completions/min_terminated_length": 342.0, "completions/max_terminated_length": 392.0, "rewards/meter/mean": 0.9989774823188782, "rewards/meter/std": 0.0006195669411681592, "rewards/count_adherence/mean": 0.7604166269302368, "rewards/count_adherence/std": 0.029462775215506554, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9926470518112183, "rewards/repeat_penalty/std": 0.020797256380319595, "rewards/total_composite/mean": 0.754129946231842, "rewards/total_composite/std": 0.03522041067481041, "reward": 0.754129946231842, "reward_std": 0.03522040322422981, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.048592861741781235, "sampling/sampling_logp_difference/max": 1.755746841430664, "sampling/importance_sampling_ratio/min": 0.1727781593799591, "sampling/importance_sampling_ratio/mean": 1.0124188661575317, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4680623523890972, "clip_ratio/low_mean": 0.019117801217362285, "clip_ratio/low_min": 0.019117801217362285, "clip_ratio/high_mean": 0.004464285913854837, "clip_ratio/high_max": 0.004464285913854837, "clip_ratio/region_mean": 0.023582087131217122, "reward_total_mean": 0.754129946231842, "reward_meter_mean": 0.9989774823188782, "reward_meter_std": 0.0006195669411681592, "reward_count_adherence_mean": 0.7604166269302368, "reward_count_adherence_std": 0.029462775215506554, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9926470518112183, "reward_repeat_penalty_std": 0.020797256380319595, "reward_total_composite_mean": 0.754129946231842, "reward_total_composite_std": 0.03522041067481041} {"timestamp_utc": "2026-04-12T03:36:54Z", "mode": "train", "global_step": 3242, "epoch": 0.1302164919468209, "loss": 0.015, "grad_norm": 2.48710560798645, "learning_rate": 1.7878787878787882e-07, "num_tokens": 7383827.0, "completions/mean_length": 275.375, "completions/min_length": 269.0, "completions/max_length": 283.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 275.375, "completions/min_terminated_length": 269.0, "completions/max_terminated_length": 283.0, "rewards/meter/mean": 0.9965729713439941, "rewards/meter/std": 0.0008044191054068506, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9411764740943909, "rewards/repeat_penalty/std": 0.03144249692559242, "rewards/total_composite/mean": 0.8207110166549683, "rewards/total_composite/std": 0.027550045400857925, "reward": 0.8207110166549683, "reward_std": 0.027550052851438522, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04147728160023689, "sampling/sampling_logp_difference/max": 1.294790267944336, "sampling/importance_sampling_ratio/min": 0.2739553153514862, "sampling/importance_sampling_ratio/mean": 1.0088013410568237, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39379945397377014, "clip_ratio/low_mean": 0.012771393987350166, "clip_ratio/low_min": 0.012771393987350166, "clip_ratio/high_mean": 0.013120089075528085, "clip_ratio/high_max": 0.013120089075528085, "clip_ratio/region_mean": 0.02589148306287825, "reward_total_mean": 0.8207110166549683, "reward_meter_mean": 0.9965729713439941, "reward_meter_std": 0.0008044191054068506, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9411764740943909, "reward_repeat_penalty_std": 0.03144249692559242, "reward_total_composite_mean": 0.8207110166549683, "reward_total_composite_std": 0.027550045400857925} {"timestamp_utc": "2026-04-12T03:37:00Z", "mode": "train", "global_step": 3243, "epoch": 0.13025665742860584, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.7575757575757576e-07, "num_tokens": 7386683.0, "completions/mean_length": 158.0, "completions/min_length": 158.0, "completions/max_length": 158.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 158.0, "completions/min_terminated_length": 158.0, "completions/max_terminated_length": 158.0, "rewards/meter/mean": 0.6563846468925476, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7272727489471436, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.47737064957618713, "rewards/total_composite/std": 0.0, "reward": 0.47737064957618713, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0011825277470052242, "sampling/sampling_logp_difference/max": 0.05302273482084274, "sampling/importance_sampling_ratio/min": 0.9712576866149902, "sampling/importance_sampling_ratio/mean": 1.0010987520217896, "sampling/importance_sampling_ratio/max": 1.0544536113739014, "entropy": 0.009368695667944849, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.47737064957618713, "reward_meter_mean": 0.6563846468925476, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7272727489471436, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.47737064957618713, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:37:05Z", "mode": "train", "global_step": 3244, "epoch": 0.1302968229103908, "loss": -0.0034, "grad_norm": 1.8808863162994385, "learning_rate": 1.7272727272727275e-07, "num_tokens": 7389632.0, "completions/mean_length": 155.625, "completions/min_length": 153.0, "completions/max_length": 159.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 155.625, "completions/min_terminated_length": 153.0, "completions/max_terminated_length": 159.0, "rewards/meter/mean": 0.998309850692749, "rewards/meter/std": 0.0023174018133431673, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998309850692749, "rewards/total_composite/std": 0.0023174018133431673, "reward": 0.998309850692749, "reward_std": 0.002317406702786684, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.039463937282562256, "sampling/sampling_logp_difference/max": 1.5400009155273438, "sampling/importance_sampling_ratio/min": 0.2143809050321579, "sampling/importance_sampling_ratio/mean": 1.0025235414505005, "sampling/importance_sampling_ratio/max": 1.5186190605163574, "entropy": 0.330673573538661, "clip_ratio/low_mean": 0.003246753243729472, "clip_ratio/low_min": 0.003246753243729472, "clip_ratio/high_mean": 0.02562888152897358, "clip_ratio/high_max": 0.02562888152897358, "clip_ratio/region_mean": 0.02887563477270305, "reward_total_mean": 0.998309850692749, "reward_meter_mean": 0.998309850692749, "reward_meter_std": 0.0023174018133431673, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.998309850692749, "reward_total_composite_std": 0.0023174018133431673} {"timestamp_utc": "2026-04-12T03:37:11Z", "mode": "train", "global_step": 3245, "epoch": 0.13033698839217575, "loss": -0.0017, "grad_norm": 2.231248378753662, "learning_rate": 1.6969696969696974e-07, "num_tokens": 7392187.0, "completions/mean_length": 135.375, "completions/min_length": 133.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 135.375, "completions/min_terminated_length": 133.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.9989526271820068, "rewards/meter/std": 0.0006340565159916878, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.9632641077041626, "rewards/total_composite/std": 0.0658845379948616, "reward": 0.9632641077041626, "reward_std": 0.0658845454454422, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02810845896601677, "sampling/sampling_logp_difference/max": 0.8493685722351074, "sampling/importance_sampling_ratio/min": 0.4276849031448364, "sampling/importance_sampling_ratio/mean": 1.0051708221435547, "sampling/importance_sampling_ratio/max": 1.8710554838180542, "entropy": 0.206793999299407, "clip_ratio/low_mean": 0.006537176435813308, "clip_ratio/low_min": 0.006537176435813308, "clip_ratio/high_mean": 0.01926107343751937, "clip_ratio/high_max": 0.01926107343751937, "clip_ratio/region_mean": 0.02579824987333268, "reward_total_mean": 0.9632641077041626, "reward_meter_mean": 0.9989526271820068, "reward_meter_std": 0.0006340565159916878, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.9632641077041626, "reward_total_composite_std": 0.0658845379948616} {"timestamp_utc": "2026-04-12T03:37:18Z", "mode": "train", "global_step": 3246, "epoch": 0.1303771538739607, "loss": 0.0198, "grad_norm": 1.8668049573898315, "learning_rate": 1.6666666666666668e-07, "num_tokens": 7395817.0, "completions/mean_length": 270.75, "completions/min_length": 259.0, "completions/max_length": 288.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 270.75, "completions/min_terminated_length": 259.0, "completions/max_terminated_length": 288.0, "rewards/meter/mean": 0.9602240324020386, "rewards/meter/std": 0.09070871770381927, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.06681530922651291, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.915900707244873, "rewards/repeat_penalty/std": 0.0559099018573761, "rewards/total_composite/mean": 0.8224383592605591, "rewards/total_composite/std": 0.09149650484323502, "reward": 0.8224383592605591, "reward_std": 0.09149649739265442, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03546025604009628, "sampling/sampling_logp_difference/max": 1.9990062713623047, "sampling/importance_sampling_ratio/min": 0.13546983897686005, "sampling/importance_sampling_ratio/mean": 1.01006281375885, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.38703763484954834, "clip_ratio/low_mean": 0.010340243112295866, "clip_ratio/low_min": 0.010340243112295866, "clip_ratio/high_mean": 0.01162644021678716, "clip_ratio/high_max": 0.01162644021678716, "clip_ratio/region_mean": 0.021966683329083025, "reward_total_mean": 0.8224383592605591, "reward_meter_mean": 0.9602240324020386, "reward_meter_std": 0.09070871770381927, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.06681530922651291, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.915900707244873, "reward_repeat_penalty_std": 0.0559099018573761, "reward_total_composite_mean": 0.8224383592605591, "reward_total_composite_std": 0.09149650484323502} {"timestamp_utc": "2026-04-12T03:37:23Z", "mode": "train", "global_step": 3247, "epoch": 0.13041731935574566, "loss": 0.0001, "grad_norm": 0.025763750076293945, "learning_rate": 1.6363636363636367e-07, "num_tokens": 7397609.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9973357319831848, "rewards/meter/std": 8.977292964118533e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973357319831848, "rewards/total_composite/std": 8.977292964118533e-06, "reward": 0.9973357319831848, "reward_std": 8.97126938070869e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0030104033648967743, "sampling/sampling_logp_difference/max": 0.4952678680419922, "sampling/importance_sampling_ratio/min": 0.6253011226654053, "sampling/importance_sampling_ratio/mean": 1.0012048482894897, "sampling/importance_sampling_ratio/max": 1.6409376859664917, "entropy": 0.012363885063678026, "clip_ratio/low_mean": 0.004098360426723957, "clip_ratio/low_min": 0.004098360426723957, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.004098360426723957, "reward_total_mean": 0.9973357319831848, "reward_meter_mean": 0.9973357319831848, "reward_meter_std": 8.977292964118533e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9973357319831848, "reward_total_composite_std": 8.977292964118533e-06} {"timestamp_utc": "2026-04-12T03:37:28Z", "mode": "train", "global_step": 3248, "epoch": 0.1304574848375306, "loss": -0.0118, "grad_norm": 4.575743675231934, "learning_rate": 1.606060606060606e-07, "num_tokens": 7400383.0, "completions/mean_length": 166.75, "completions/min_length": 159.0, "completions/max_length": 173.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 166.75, "completions/min_terminated_length": 159.0, "completions/max_terminated_length": 173.0, "rewards/meter/mean": 0.9989683628082275, "rewards/meter/std": 0.00019462761702015996, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.9712226390838623, "rewards/total_composite/std": 0.051451049745082855, "reward": 0.9712226390838623, "reward_std": 0.051451049745082855, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.040822967886924744, "sampling/sampling_logp_difference/max": 1.530099868774414, "sampling/importance_sampling_ratio/min": 0.21651406586170197, "sampling/importance_sampling_ratio/mean": 1.0083783864974976, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3529275842010975, "clip_ratio/low_mean": 0.0085039883852005, "clip_ratio/low_min": 0.0085039883852005, "clip_ratio/high_mean": 0.023729265900328755, "clip_ratio/high_max": 0.023729265900328755, "clip_ratio/region_mean": 0.032233254285529256, "reward_total_mean": 0.9712226390838623, "reward_meter_mean": 0.9989683628082275, "reward_meter_std": 0.00019462761702015996, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_total_composite_mean": 0.9712226390838623, "reward_total_composite_std": 0.051451049745082855} {"timestamp_utc": "2026-04-12T03:37:39Z", "mode": "train", "global_step": 3249, "epoch": 0.13049765031931557, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.575757575757576e-07, "num_tokens": 7402303.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.998656690120697, "rewards/meter/std": 0.0011350500863045454, "rewards/count_adherence/mean": 0.7291666269302368, "rewards/count_adherence/std": 0.01964186504483223, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9807692170143127, "rewards/repeat_penalty/std": 0.02907419577240944, "rewards/total_composite/mean": 0.7142915725708008, "rewards/total_composite/std": 0.03164464980363846, "reward": 0.7142915725708008, "reward_std": 0.03164464607834816, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.7142915725708008, "reward_meter_mean": 0.998656690120697, "reward_meter_std": 0.0011350500863045454, "reward_count_adherence_mean": 0.7291666269302368, "reward_count_adherence_std": 0.01964186504483223, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9807692170143127, "reward_repeat_penalty_std": 0.02907419577240944, "reward_total_composite_mean": 0.7142915725708008, "reward_total_composite_std": 0.03164464980363846} {"timestamp_utc": "2026-04-12T03:37:44Z", "mode": "train", "global_step": 3250, "epoch": 0.13053781580110052, "loss": -0.0012, "grad_norm": 1.1469806432724, "learning_rate": 1.5454545454545456e-07, "num_tokens": 7404902.0, "completions/mean_length": 141.875, "completions/min_length": 140.0, "completions/max_length": 143.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 141.875, "completions/min_terminated_length": 140.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.9991378784179688, "rewards/meter/std": 0.00011923787678824738, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991378784179688, "rewards/total_composite/std": 0.00011923787678824738, "reward": 0.9991378784179688, "reward_std": 0.00011922277917619795, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020554685965180397, "sampling/sampling_logp_difference/max": 1.2523746490478516, "sampling/importance_sampling_ratio/min": 0.285825252532959, "sampling/importance_sampling_ratio/mean": 1.0037734508514404, "sampling/importance_sampling_ratio/max": 1.4649658203125, "entropy": 0.20848924107849598, "clip_ratio/low_mean": 0.010569972451776266, "clip_ratio/low_min": 0.010569972451776266, "clip_ratio/high_mean": 0.00882849539630115, "clip_ratio/high_max": 0.00882849539630115, "clip_ratio/region_mean": 0.019398467848077416, "reward_total_mean": 0.9991378784179688, "reward_meter_mean": 0.9991378784179688, "reward_meter_std": 0.00011923787678824738, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9991378784179688, "reward_total_composite_std": 0.00011923787678824738} {"timestamp_utc": "2026-04-12T03:39:02Z", "mode": "eval", "global_step": 3250, "epoch": 0.13053781580110052, "eval_loss": NaN, "eval_runtime": 78.0663, "eval_samples_per_second": 1.332, "eval_steps_per_second": 0.167, "eval_num_tokens": 7404902.0, "eval_completions/mean_length": 214.33653846153845, "eval_completions/min_length": 60.76923076923077, "eval_completions/max_length": 421.38461538461536, "eval_completions/clipped_ratio": 0.028846153846153848, "eval_completions/mean_terminated_length": 204.00686880258414, "eval_completions/min_terminated_length": 60.76923076923077, "eval_completions/max_terminated_length": 398.7692307692308, "eval_rewards/meter/mean": 0.7977132155345037, "eval_rewards/meter/std": 0.3206370784542881, "eval_rewards/count_adherence/mean": 0.95811671935595, "eval_rewards/count_adherence/std": 0.06441009388520168, "eval_rewards/arabic_clean/mean": 0.9807692307692307, "eval_rewards/arabic_clean/std": 0.05439282839114849, "eval_rewards/repeat_penalty/mean": 0.9612986537126395, "eval_rewards/repeat_penalty/std": 0.06227556343835134, "eval_rewards/total_composite/mean": 0.7313130910579975, "eval_rewards/total_composite/std": 0.33714570047763676, "eval_reward": 0.7313130910579975, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.034000108448358685, "eval_sampling/sampling_logp_difference/max": 1.2143818781926081, "eval_sampling/importance_sampling_ratio/min": 0.30588124233942765, "eval_sampling/importance_sampling_ratio/mean": 1.0085856593572176, "eval_sampling/importance_sampling_ratio/max": 1.4786042525218084, "eval_entropy": 0.38327448299297917, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7313130910579975, "eval_reward_meter_mean": 0.7977132155345037, "eval_reward_meter_std": 0.3206370784542881, "eval_reward_count_adherence_mean": 0.95811671935595, "eval_reward_count_adherence_std": 0.06441009388520168, "eval_reward_arabic_clean_mean": 0.9807692307692307, "eval_reward_arabic_clean_std": 0.05439282839114849, "eval_reward_repeat_penalty_mean": 0.9612986537126395, "eval_reward_repeat_penalty_std": 0.06227556343835134, "eval_reward_total_composite_mean": 0.7313130910579975, "eval_reward_total_composite_std": 0.33714570047763676} {"timestamp_utc": "2026-04-12T03:39:09Z", "mode": "train", "global_step": 3251, "epoch": 0.13057798128288547, "loss": 0.003, "grad_norm": 1.9768705368041992, "learning_rate": 1.5151515151515152e-07, "num_tokens": 7406604.0, "completions/mean_length": 40.75, "completions/min_length": 39.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.75, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.998726487159729, "rewards/meter/std": 0.0001513757451903075, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998726487159729, "rewards/total_composite/std": 0.0001513757451903075, "reward": 0.998726487159729, "reward_std": 0.0001513654860900715, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020130213350057602, "sampling/sampling_logp_difference/max": 1.2893218994140625, "sampling/importance_sampling_ratio/min": 0.275457501411438, "sampling/importance_sampling_ratio/mean": 0.9942156076431274, "sampling/importance_sampling_ratio/max": 1.5404958724975586, "entropy": 0.12377766706049442, "clip_ratio/low_mean": 0.006253908621147275, "clip_ratio/low_min": 0.006253908621147275, "clip_ratio/high_mean": 0.009146341122686863, "clip_ratio/high_max": 0.009146341122686863, "clip_ratio/region_mean": 0.015400249743834138, "reward_total_mean": 0.998726487159729, "reward_meter_mean": 0.998726487159729, "reward_meter_std": 0.0001513757451903075, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.998726487159729, "reward_total_composite_std": 0.0001513757451903075} {"timestamp_utc": "2026-04-12T03:39:15Z", "mode": "train", "global_step": 3252, "epoch": 0.13061814676467043, "loss": 0.0033, "grad_norm": 3.5564746856689453, "learning_rate": 1.484848484848485e-07, "num_tokens": 7409050.0, "completions/mean_length": 124.75, "completions/min_length": 123.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 124.75, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.9973940253257751, "rewards/meter/std": 0.0014456679346039891, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973940253257751, "rewards/total_composite/std": 0.0014456679346039891, "reward": 0.9973940253257751, "reward_std": 0.001445673406124115, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018116910010576248, "sampling/sampling_logp_difference/max": 1.4764413833618164, "sampling/importance_sampling_ratio/min": 0.22844921052455902, "sampling/importance_sampling_ratio/mean": 1.0002362728118896, "sampling/importance_sampling_ratio/max": 1.582746148109436, "entropy": 0.12918496038764715, "clip_ratio/low_mean": 0.0009920635493472219, "clip_ratio/low_min": 0.0009920635493472219, "clip_ratio/high_mean": 0.012048649485222995, "clip_ratio/high_max": 0.012048649485222995, "clip_ratio/region_mean": 0.013040713034570217, "reward_total_mean": 0.9973940253257751, "reward_meter_mean": 0.9973940253257751, "reward_meter_std": 0.0014456679346039891, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9973940253257751, "reward_total_composite_std": 0.0014456679346039891} {"timestamp_utc": "2026-04-12T03:39:21Z", "mode": "train", "global_step": 3253, "epoch": 0.13065831224645538, "loss": 0.008, "grad_norm": 1.9089981317520142, "learning_rate": 1.4545454545454548e-07, "num_tokens": 7411813.0, "completions/mean_length": 154.375, "completions/min_length": 152.0, "completions/max_length": 156.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 154.375, "completions/min_terminated_length": 152.0, "completions/max_terminated_length": 156.0, "rewards/meter/mean": 0.9973270297050476, "rewards/meter/std": 0.0007912821019999683, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9861111044883728, "rewards/repeat_penalty/std": 0.03928370773792267, "rewards/total_composite/mean": 0.983467161655426, "rewards/total_composite/std": 0.038973402231931686, "reward": 0.983467161655426, "reward_std": 0.03897340968251228, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.029868151992559433, "sampling/sampling_logp_difference/max": 1.405970573425293, "sampling/importance_sampling_ratio/min": 0.24512901902198792, "sampling/importance_sampling_ratio/mean": 1.0039182901382446, "sampling/importance_sampling_ratio/max": 1.7957075834274292, "entropy": 0.2324562668800354, "clip_ratio/low_mean": 0.002419354859739542, "clip_ratio/low_min": 0.002419354859739542, "clip_ratio/high_mean": 0.017753356718458235, "clip_ratio/high_max": 0.017753356718458235, "clip_ratio/region_mean": 0.020172711578197777, "reward_total_mean": 0.983467161655426, "reward_meter_mean": 0.9973270297050476, "reward_meter_std": 0.0007912821019999683, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9861111044883728, "reward_repeat_penalty_std": 0.03928370773792267, "reward_total_composite_mean": 0.983467161655426, "reward_total_composite_std": 0.038973402231931686} {"timestamp_utc": "2026-04-12T03:39:28Z", "mode": "train", "global_step": 3254, "epoch": 0.13069847772824034, "loss": -0.0002, "grad_norm": 2.1981167793273926, "learning_rate": 1.4242424242424244e-07, "num_tokens": 7415280.0, "completions/mean_length": 232.375, "completions/min_length": 227.0, "completions/max_length": 241.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 232.375, "completions/min_terminated_length": 227.0, "completions/max_terminated_length": 241.0, "rewards/meter/mean": 0.9978033304214478, "rewards/meter/std": 0.0022864267230033875, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9807692170143127, "rewards/repeat_penalty/std": 0.03560846298933029, "rewards/total_composite/mean": 0.978593111038208, "rewards/total_composite/std": 0.03494200482964516, "reward": 0.978593111038208, "reward_std": 0.03494199737906456, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.042390793561935425, "sampling/sampling_logp_difference/max": 1.2398028373718262, "sampling/importance_sampling_ratio/min": 0.2894412875175476, "sampling/importance_sampling_ratio/mean": 1.0061860084533691, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.34237322956323624, "clip_ratio/low_mean": 0.007630055071786046, "clip_ratio/low_min": 0.007630055071786046, "clip_ratio/high_mean": 0.025140726240351796, "clip_ratio/high_max": 0.025140726240351796, "clip_ratio/region_mean": 0.03277078131213784, "reward_total_mean": 0.978593111038208, "reward_meter_mean": 0.9978033304214478, "reward_meter_std": 0.0022864267230033875, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9807692170143127, "reward_repeat_penalty_std": 0.03560846298933029, "reward_total_composite_mean": 0.978593111038208, "reward_total_composite_std": 0.03494200482964516} {"timestamp_utc": "2026-04-12T03:39:33Z", "mode": "train", "global_step": 3255, "epoch": 0.1307386432100253, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.393939393939394e-07, "num_tokens": 7417000.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 8.257458102889359e-05, "sampling/sampling_logp_difference/max": 0.005363806616514921, "sampling/importance_sampling_ratio/min": 0.9946505427360535, "sampling/importance_sampling_ratio/mean": 1.0000584125518799, "sampling/importance_sampling_ratio/max": 1.0015891790390015, "entropy": 0.000789376012107823, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:39:38Z", "mode": "train", "global_step": 3256, "epoch": 0.13077880869181027, "loss": 0.0005, "grad_norm": 1.5352351665496826, "learning_rate": 1.3636363636363637e-07, "num_tokens": 7419231.0, "completions/mean_length": 100.875, "completions/min_length": 100.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.875, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.9993561506271362, "rewards/meter/std": 0.00013285702152643353, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9743703603744507, "rewards/total_composite/std": 0.07064089924097061, "reward": 0.9743703603744507, "reward_std": 0.07064089179039001, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01845725066959858, "sampling/sampling_logp_difference/max": 1.2666641473770142, "sampling/importance_sampling_ratio/min": 0.2817700207233429, "sampling/importance_sampling_ratio/mean": 1.0004955530166626, "sampling/importance_sampling_ratio/max": 1.6758545637130737, "entropy": 0.13637553714215755, "clip_ratio/low_mean": 0.0012376237427815795, "clip_ratio/low_min": 0.0012376237427815795, "clip_ratio/high_mean": 0.01858959416858852, "clip_ratio/high_max": 0.01858959416858852, "clip_ratio/region_mean": 0.0198272179113701, "reward_total_mean": 0.9743703603744507, "reward_meter_mean": 0.9993561506271362, "reward_meter_std": 0.00013285702152643353, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9743703603744507, "reward_total_composite_std": 0.07064089924097061} {"timestamp_utc": "2026-04-12T03:39:42Z", "mode": "train", "global_step": 3257, "epoch": 0.13081897417359523, "loss": 0.0003, "grad_norm": 1.7167513370513916, "learning_rate": 1.3333333333333336e-07, "num_tokens": 7421119.0, "completions/mean_length": 71.0, "completions/min_length": 70.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.0, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9994316101074219, "rewards/meter/std": 0.0001929309801198542, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994316101074219, "rewards/total_composite/std": 0.0001929309801198542, "reward": 0.9994316101074219, "reward_std": 0.00019294493540655822, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014849050901830196, "sampling/sampling_logp_difference/max": 1.1428728103637695, "sampling/importance_sampling_ratio/min": 0.31890156865119934, "sampling/importance_sampling_ratio/mean": 0.9987840056419373, "sampling/importance_sampling_ratio/max": 1.2882606983184814, "entropy": 0.0983225479722023, "clip_ratio/low_mean": 0.007067404338158667, "clip_ratio/low_min": 0.007067404338158667, "clip_ratio/high_mean": 0.008754611015319824, "clip_ratio/high_max": 0.008754611015319824, "clip_ratio/region_mean": 0.01582201535347849, "reward_total_mean": 0.9994316101074219, "reward_meter_mean": 0.9994316101074219, "reward_meter_std": 0.0001929309801198542, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994316101074219, "reward_total_composite_std": 0.0001929309801198542} {"timestamp_utc": "2026-04-12T03:39:47Z", "mode": "train", "global_step": 3258, "epoch": 0.13085913965538018, "loss": 0.0004, "grad_norm": 0.16614317893981934, "learning_rate": 1.3030303030303033e-07, "num_tokens": 7422983.0, "completions/mean_length": 71.0, "completions/min_length": 71.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.0, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9994429349899292, "rewards/meter/std": 1.4887036741129123e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994429349899292, "rewards/total_composite/std": 1.4887036741129123e-05, "reward": 0.9994429349899292, "reward_std": 1.4891715181875043e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013734190724790096, "sampling/sampling_logp_difference/max": 1.3592395782470703, "sampling/importance_sampling_ratio/min": 0.2568560242652893, "sampling/importance_sampling_ratio/mean": 0.9997973442077637, "sampling/importance_sampling_ratio/max": 1.2386420965194702, "entropy": 0.0794384153559804, "clip_ratio/low_mean": 0.0017605633474886417, "clip_ratio/low_min": 0.0017605633474886417, "clip_ratio/high_mean": 0.01056338008493185, "clip_ratio/high_max": 0.01056338008493185, "clip_ratio/region_mean": 0.012323943432420492, "reward_total_mean": 0.9994429349899292, "reward_meter_mean": 0.9994429349899292, "reward_meter_std": 1.4887036741129123e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994429349899292, "reward_total_composite_std": 1.4887036741129123e-05} {"timestamp_utc": "2026-04-12T03:39:53Z", "mode": "train", "global_step": 3259, "epoch": 0.13089930513716513, "loss": -0.0021, "grad_norm": 6.062583923339844, "learning_rate": 1.272727272727273e-07, "num_tokens": 7424765.0, "completions/mean_length": 65.75, "completions/min_length": 63.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.75, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9497079849243164, "rewards/meter/std": 0.005244073923677206, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9497079849243164, "rewards/total_composite/std": 0.005244073923677206, "reward": 0.9497079849243164, "reward_std": 0.00524406973272562, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030090492218732834, "sampling/sampling_logp_difference/max": 1.032027244567871, "sampling/importance_sampling_ratio/min": 0.3562839925289154, "sampling/importance_sampling_ratio/mean": 1.0019686222076416, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1472517903894186, "clip_ratio/low_mean": 0.01923532225191593, "clip_ratio/low_min": 0.01923532225191593, "clip_ratio/high_mean": 0.018950162804685533, "clip_ratio/high_max": 0.018950162804685533, "clip_ratio/region_mean": 0.038185485056601465, "reward_total_mean": 0.9497079849243164, "reward_meter_mean": 0.9497079849243164, "reward_meter_std": 0.005244073923677206, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9497079849243164, "reward_total_composite_std": 0.005244073923677206} {"timestamp_utc": "2026-04-12T03:39:57Z", "mode": "train", "global_step": 3260, "epoch": 0.1309394706189501, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.2424242424242426e-07, "num_tokens": 7426277.0, "completions/mean_length": 32.0, "completions/min_length": 32.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "reward": 0.9992982149124146, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 5.620270167128183e-05, "sampling/sampling_logp_difference/max": 0.001364716561511159, "sampling/importance_sampling_ratio/min": 0.9986362457275391, "sampling/importance_sampling_ratio/mean": 1.0000357627868652, "sampling/importance_sampling_ratio/max": 1.0006037950515747, "entropy": 0.0003866630977427121, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9992982149124146, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:40:01Z", "mode": "train", "global_step": 3261, "epoch": 0.13097963610073504, "loss": 0.0044, "grad_norm": 3.223155975341797, "learning_rate": 1.2121212121212122e-07, "num_tokens": 7428154.0, "completions/mean_length": 69.625, "completions/min_length": 66.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.625, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9926314949989319, "rewards/meter/std": 0.006077465135604143, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9926314949989319, "rewards/total_composite/std": 0.006077465135604143, "reward": 0.9926314949989319, "reward_std": 0.006077463272958994, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.029856102541089058, "sampling/sampling_logp_difference/max": 1.1406583786010742, "sampling/importance_sampling_ratio/min": 0.31960853934288025, "sampling/importance_sampling_ratio/mean": 1.0040087699890137, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1911148466169834, "clip_ratio/low_mean": 0.007126289419829845, "clip_ratio/low_min": 0.007126289419829845, "clip_ratio/high_mean": 0.012933629681356251, "clip_ratio/high_max": 0.012933629681356251, "clip_ratio/region_mean": 0.020059919101186097, "reward_total_mean": 0.9926314949989319, "reward_meter_mean": 0.9926314949989319, "reward_meter_std": 0.006077465135604143, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9926314949989319, "reward_total_composite_std": 0.006077465135604143} {"timestamp_utc": "2026-04-12T03:40:06Z", "mode": "train", "global_step": 3262, "epoch": 0.13101980158252, "loss": 0.0013, "grad_norm": 0.49830862879753113, "learning_rate": 1.1818181818181818e-07, "num_tokens": 7429891.0, "completions/mean_length": 71.125, "completions/min_length": 71.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.125, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9994339346885681, "rewards/meter/std": 2.7899814085685648e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994339346885681, "rewards/total_composite/std": 2.7899814085685648e-05, "reward": 0.9994339346885681, "reward_std": 2.791790757328272e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011753112077713013, "sampling/sampling_logp_difference/max": 1.6477203369140625, "sampling/importance_sampling_ratio/min": 0.19248820841312408, "sampling/importance_sampling_ratio/mean": 1.0001577138900757, "sampling/importance_sampling_ratio/max": 1.19877290725708, "entropy": 0.07193375751376152, "clip_ratio/low_mean": 0.0034966744715347886, "clip_ratio/low_min": 0.0034966744715347886, "clip_ratio/high_mean": 0.0017605633474886417, "clip_ratio/high_max": 0.0017605633474886417, "clip_ratio/region_mean": 0.00525723781902343, "reward_total_mean": 0.9994339346885681, "reward_meter_mean": 0.9994339346885681, "reward_meter_std": 2.7899814085685648e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994339346885681, "reward_total_composite_std": 2.7899814085685648e-05} {"timestamp_utc": "2026-04-12T03:40:11Z", "mode": "train", "global_step": 3263, "epoch": 0.13105996706430495, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.1515151515151516e-07, "num_tokens": 7431611.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9973388910293579, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973388910293579, "rewards/total_composite/std": 0.0, "reward": 0.9973388910293579, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0008966223103925586, "sampling/sampling_logp_difference/max": 0.09105479717254639, "sampling/importance_sampling_ratio/min": 0.9129676818847656, "sampling/importance_sampling_ratio/mean": 1.0003302097320557, "sampling/importance_sampling_ratio/max": 1.0381407737731934, "entropy": 0.012014997191727161, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9973388910293579, "reward_meter_mean": 0.9973388910293579, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9973388910293579, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:40:16Z", "mode": "train", "global_step": 3264, "epoch": 0.1311001325460899, "loss": 0.0001, "grad_norm": 0.04755792394280434, "learning_rate": 1.1212121212121213e-07, "num_tokens": 7433682.0, "completions/mean_length": 97.875, "completions/min_length": 97.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.875, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.999409019947052, "rewards/meter/std": 5.810459242638899e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999409019947052, "rewards/total_composite/std": 5.810459242638899e-06, "reward": 0.999409019947052, "reward_std": 5.812751624034718e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003859634278342128, "sampling/sampling_logp_difference/max": 0.3599740266799927, "sampling/importance_sampling_ratio/min": 0.7912850379943848, "sampling/importance_sampling_ratio/mean": 1.0018383264541626, "sampling/importance_sampling_ratio/max": 1.4332921504974365, "entropy": 0.03584319236688316, "clip_ratio/low_mean": 0.0038265305338427424, "clip_ratio/low_min": 0.0038265305338427424, "clip_ratio/high_mean": 0.003839680110104382, "clip_ratio/high_max": 0.003839680110104382, "clip_ratio/region_mean": 0.0076662106439471245, "reward_total_mean": 0.999409019947052, "reward_meter_mean": 0.999409019947052, "reward_meter_std": 5.810459242638899e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.999409019947052, "reward_total_composite_std": 5.810459242638899e-06} {"timestamp_utc": "2026-04-12T03:40:22Z", "mode": "train", "global_step": 3265, "epoch": 0.13114029802787486, "loss": 0.0039, "grad_norm": 2.038606643676758, "learning_rate": 1.090909090909091e-07, "num_tokens": 7436489.0, "completions/mean_length": 166.875, "completions/min_length": 162.0, "completions/max_length": 169.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 166.875, "completions/min_terminated_length": 162.0, "completions/max_terminated_length": 169.0, "rewards/meter/mean": 0.999045729637146, "rewards/meter/std": 0.00020887328719254583, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999045729637146, "rewards/total_composite/std": 0.00020887328719254583, "reward": 0.999045729637146, "reward_std": 0.00020887046412099153, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03505001217126846, "sampling/sampling_logp_difference/max": 1.5322790145874023, "sampling/importance_sampling_ratio/min": 0.21604275703430176, "sampling/importance_sampling_ratio/mean": 1.0036981105804443, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31825447641313076, "clip_ratio/low_mean": 0.011210258584469557, "clip_ratio/low_min": 0.011210258584469557, "clip_ratio/high_mean": 0.015736370929516852, "clip_ratio/high_max": 0.015736370929516852, "clip_ratio/region_mean": 0.02694662951398641, "reward_total_mean": 0.999045729637146, "reward_meter_mean": 0.999045729637146, "reward_meter_std": 0.00020887328719254583, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.999045729637146, "reward_total_composite_std": 0.00020887328719254583} {"timestamp_utc": "2026-04-12T03:40:28Z", "mode": "train", "global_step": 3266, "epoch": 0.1311804635096598, "loss": -0.003, "grad_norm": 1.5206332206726074, "learning_rate": 1.0606060606060608e-07, "num_tokens": 7439156.0, "completions/mean_length": 155.375, "completions/min_length": 152.0, "completions/max_length": 159.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 155.375, "completions/min_terminated_length": 152.0, "completions/max_terminated_length": 159.0, "rewards/meter/mean": 0.9990906119346619, "rewards/meter/std": 0.0002633438853081316, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990906119346619, "rewards/total_composite/std": 0.0002633438853081316, "reward": 0.9990906119346619, "reward_std": 0.0002633513940963894, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031446803361177444, "sampling/sampling_logp_difference/max": 1.307551383972168, "sampling/importance_sampling_ratio/min": 0.27048155665397644, "sampling/importance_sampling_ratio/mean": 1.009647250175476, "sampling/importance_sampling_ratio/max": 1.7514793872833252, "entropy": 0.30485600233078003, "clip_ratio/low_mean": 0.003257433301769197, "clip_ratio/low_min": 0.003257433301769197, "clip_ratio/high_mean": 0.018389825709164143, "clip_ratio/high_max": 0.018389825709164143, "clip_ratio/region_mean": 0.02164725901093334, "reward_total_mean": 0.9990906119346619, "reward_meter_mean": 0.9990906119346619, "reward_meter_std": 0.0002633438853081316, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990906119346619, "reward_total_composite_std": 0.0002633438853081316} {"timestamp_utc": "2026-04-12T03:40:33Z", "mode": "train", "global_step": 3267, "epoch": 0.13122062899144477, "loss": 0.0021, "grad_norm": 1.932774305343628, "learning_rate": 1.0303030303030304e-07, "num_tokens": 7441984.0, "completions/mean_length": 156.5, "completions/min_length": 155.0, "completions/max_length": 159.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 156.5, "completions/min_terminated_length": 155.0, "completions/max_terminated_length": 159.0, "rewards/meter/mean": 0.9990622997283936, "rewards/meter/std": 0.00025987729895859957, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990622997283936, "rewards/total_composite/std": 0.00025987729895859957, "reward": 0.9990622997283936, "reward_std": 0.0002598702849354595, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.035560522228479385, "sampling/sampling_logp_difference/max": 1.2120389938354492, "sampling/importance_sampling_ratio/min": 0.29758986830711365, "sampling/importance_sampling_ratio/mean": 1.0105011463165283, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3256886452436447, "clip_ratio/low_mean": 0.004726932500489056, "clip_ratio/low_min": 0.004726932500489056, "clip_ratio/high_mean": 0.011238692968618125, "clip_ratio/high_max": 0.011238692968618125, "clip_ratio/region_mean": 0.01596562546910718, "reward_total_mean": 0.9990622997283936, "reward_meter_mean": 0.9990622997283936, "reward_meter_std": 0.00025987729895859957, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990622997283936, "reward_total_composite_std": 0.00025987729895859957} {"timestamp_utc": "2026-04-12T03:40:39Z", "mode": "train", "global_step": 3268, "epoch": 0.13126079447322972, "loss": -0.0011, "grad_norm": 0.6131901741027832, "learning_rate": 1.0000000000000001e-07, "num_tokens": 7444602.0, "completions/mean_length": 131.25, "completions/min_length": 130.0, "completions/max_length": 132.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.25, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.9992755651473999, "rewards/meter/std": 0.00033690276904962957, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992755651473999, "rewards/total_composite/std": 0.00033690276904962957, "reward": 0.9992755651473999, "reward_std": 0.0003369078040122986, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011317918077111244, "sampling/sampling_logp_difference/max": 2.1706619262695312, "sampling/importance_sampling_ratio/min": 0.1141020655632019, "sampling/importance_sampling_ratio/mean": 0.9999168515205383, "sampling/importance_sampling_ratio/max": 1.4053065776824951, "entropy": 0.08114278595894575, "clip_ratio/low_mean": 0.0019083969527855515, "clip_ratio/low_min": 0.0019083969527855515, "clip_ratio/high_mean": 0.005740093358326703, "clip_ratio/high_max": 0.005740093358326703, "clip_ratio/region_mean": 0.007648490311112255, "reward_total_mean": 0.9992755651473999, "reward_meter_mean": 0.9992755651473999, "reward_meter_std": 0.00033690276904962957, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992755651473999, "reward_total_composite_std": 0.00033690276904962957} {"timestamp_utc": "2026-04-12T03:40:43Z", "mode": "train", "global_step": 3269, "epoch": 0.13130095995501467, "loss": 0.0089, "grad_norm": 3.9373042583465576, "learning_rate": 9.696969696969697e-08, "num_tokens": 7446314.0, "completions/mean_length": 46.0, "completions/min_length": 46.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.0, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.9440339803695679, "rewards/meter/std": 0.0030637700110673904, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9440339803695679, "rewards/total_composite/std": 0.0030637700110673904, "reward": 0.9440339803695679, "reward_std": 0.0030637700110673904, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020654289051890373, "sampling/sampling_logp_difference/max": 1.6016666889190674, "sampling/importance_sampling_ratio/min": 0.20156030356884003, "sampling/importance_sampling_ratio/mean": 1.0039082765579224, "sampling/importance_sampling_ratio/max": 1.5834907293319702, "entropy": 0.08023244515061378, "clip_ratio/low_mean": 0.00815217406488955, "clip_ratio/low_min": 0.00815217406488955, "clip_ratio/high_mean": 0.00815217406488955, "clip_ratio/high_max": 0.00815217406488955, "clip_ratio/region_mean": 0.0163043481297791, "reward_total_mean": 0.9440339803695679, "reward_meter_mean": 0.9440339803695679, "reward_meter_std": 0.0030637700110673904, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9440339803695679, "reward_total_composite_std": 0.0030637700110673904} {"timestamp_utc": "2026-04-12T03:40:50Z", "mode": "train", "global_step": 3270, "epoch": 0.13134112543679963, "loss": 0.0097, "grad_norm": 2.0776824951171875, "learning_rate": 9.393939393939395e-08, "num_tokens": 7449775.0, "completions/mean_length": 232.625, "completions/min_length": 226.0, "completions/max_length": 238.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 232.625, "completions/min_terminated_length": 226.0, "completions/max_terminated_length": 238.0, "rewards/meter/mean": 0.9988599419593811, "rewards/meter/std": 0.0006400212296284735, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9988599419593811, "rewards/total_composite/std": 0.0006400212296284735, "reward": 0.9988599419593811, "reward_std": 0.0006400320562534034, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04505442455410957, "sampling/sampling_logp_difference/max": 1.3682456016540527, "sampling/importance_sampling_ratio/min": 0.2545531392097473, "sampling/importance_sampling_ratio/mean": 1.0114996433258057, "sampling/importance_sampling_ratio/max": 1.781883716583252, "entropy": 0.4155345596373081, "clip_ratio/low_mean": 0.005398246110416949, "clip_ratio/low_min": 0.005398246110416949, "clip_ratio/high_mean": 0.016811800538562238, "clip_ratio/high_max": 0.016811800538562238, "clip_ratio/region_mean": 0.022210046648979187, "reward_total_mean": 0.9988599419593811, "reward_meter_mean": 0.9988599419593811, "reward_meter_std": 0.0006400212296284735, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9988599419593811, "reward_total_composite_std": 0.0006400212296284735} {"timestamp_utc": "2026-04-12T03:40:55Z", "mode": "train", "global_step": 3271, "epoch": 0.13138129091858458, "loss": 0.0018, "grad_norm": 0.9732382893562317, "learning_rate": 9.090909090909091e-08, "num_tokens": 7451918.0, "completions/mean_length": 106.875, "completions/min_length": 106.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.875, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.9993022680282593, "rewards/meter/std": 8.71433803695254e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993022680282593, "rewards/total_composite/std": 8.71433803695254e-05, "reward": 0.9993022680282593, "reward_std": 8.713654096936807e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016620170325040817, "sampling/sampling_logp_difference/max": 1.211838722229004, "sampling/importance_sampling_ratio/min": 0.29764947295188904, "sampling/importance_sampling_ratio/mean": 1.00331711769104, "sampling/importance_sampling_ratio/max": 1.4888734817504883, "entropy": 0.16768567822873592, "clip_ratio/low_mean": 0.004673101240769029, "clip_ratio/low_min": 0.004673101240769029, "clip_ratio/high_mean": 0.0093678361736238, "clip_ratio/high_max": 0.0093678361736238, "clip_ratio/region_mean": 0.014040937414392829, "reward_total_mean": 0.9993022680282593, "reward_meter_mean": 0.9993022680282593, "reward_meter_std": 8.71433803695254e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993022680282593, "reward_total_composite_std": 8.71433803695254e-05} {"timestamp_utc": "2026-04-12T03:41:01Z", "mode": "train", "global_step": 3272, "epoch": 0.13142145640036954, "loss": -0.0028, "grad_norm": 2.9168341159820557, "learning_rate": 8.787878787878788e-08, "num_tokens": 7454434.0, "completions/mean_length": 136.5, "completions/min_length": 132.0, "completions/max_length": 144.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 136.5, "completions/min_terminated_length": 132.0, "completions/max_terminated_length": 144.0, "rewards/meter/mean": 0.9926499128341675, "rewards/meter/std": 0.0036937205586582422, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9926499128341675, "rewards/total_composite/std": 0.0036937205586582422, "reward": 0.9926499128341675, "reward_std": 0.0036937135737389326, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0317571647465229, "sampling/sampling_logp_difference/max": 1.151773452758789, "sampling/importance_sampling_ratio/min": 0.31607571244239807, "sampling/importance_sampling_ratio/mean": 1.0109081268310547, "sampling/importance_sampling_ratio/max": 1.8989595174789429, "entropy": 0.310006458312273, "clip_ratio/low_mean": 0.008297410036902875, "clip_ratio/low_min": 0.008297410036902875, "clip_ratio/high_mean": 0.017071759328246117, "clip_ratio/high_max": 0.017071759328246117, "clip_ratio/region_mean": 0.02536916936514899, "reward_total_mean": 0.9926499128341675, "reward_meter_mean": 0.9926499128341675, "reward_meter_std": 0.0036937205586582422, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9926499128341675, "reward_total_composite_std": 0.0036937205586582422} {"timestamp_utc": "2026-04-12T03:41:05Z", "mode": "train", "global_step": 3273, "epoch": 0.1314616218821545, "loss": 0.0013, "grad_norm": 2.5891594886779785, "learning_rate": 8.484848484848487e-08, "num_tokens": 7456266.0, "completions/mean_length": 67.0, "completions/min_length": 67.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981439113616943, "rewards/meter/std": 3.332228152430616e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981439113616943, "rewards/total_composite/std": 3.332228152430616e-05, "reward": 0.9981439113616943, "reward_std": 3.332816413603723e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0074036987498402596, "sampling/sampling_logp_difference/max": 0.8070247173309326, "sampling/importance_sampling_ratio/min": 0.4461836516857147, "sampling/importance_sampling_ratio/mean": 0.9988389611244202, "sampling/importance_sampling_ratio/max": 1.3197448253631592, "entropy": 0.04388477047905326, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.005597014795057476, "clip_ratio/high_max": 0.005597014795057476, "clip_ratio/region_mean": 0.005597014795057476, "reward_total_mean": 0.9981439113616943, "reward_meter_mean": 0.9981439113616943, "reward_meter_std": 3.332228152430616e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981439113616943, "reward_total_composite_std": 3.332228152430616e-05} {"timestamp_utc": "2026-04-12T03:41:10Z", "mode": "train", "global_step": 3274, "epoch": 0.13150178736393944, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 8.181818181818183e-08, "num_tokens": 7458170.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00010550412116572261, "sampling/sampling_logp_difference/max": 0.005598756484687328, "sampling/importance_sampling_ratio/min": 0.9944168329238892, "sampling/importance_sampling_ratio/mean": 1.0000689029693604, "sampling/importance_sampling_ratio/max": 1.0018798112869263, "entropy": 0.0009244889370165765, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:41:14Z", "mode": "train", "global_step": 3275, "epoch": 0.1315419528457244, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.87878787878788e-08, "num_tokens": 7459834.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00023801384668331593, "sampling/sampling_logp_difference/max": 0.00208944920450449, "sampling/importance_sampling_ratio/min": 0.9997556209564209, "sampling/importance_sampling_ratio/mean": 1.0002355575561523, "sampling/importance_sampling_ratio/max": 1.002091646194458, "entropy": 0.0017418517236365005, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:41:19Z", "mode": "train", "global_step": 3276, "epoch": 0.13158211832750935, "loss": -0.0024, "grad_norm": 1.703597068786621, "learning_rate": 7.575757575757576e-08, "num_tokens": 7461675.0, "completions/mean_length": 68.125, "completions/min_length": 67.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.125, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.999450147151947, "rewards/meter/std": 0.00012958116712979972, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999450147151947, "rewards/total_composite/std": 0.00012958116712979972, "reward": 0.999450147151947, "reward_std": 0.000129588384879753, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011213197372853756, "sampling/sampling_logp_difference/max": 0.7290372848510742, "sampling/importance_sampling_ratio/min": 0.4823731780052185, "sampling/importance_sampling_ratio/mean": 1.0002046823501587, "sampling/importance_sampling_ratio/max": 1.684713363647461, "entropy": 0.06052346946671605, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.999450147151947, "reward_meter_mean": 0.999450147151947, "reward_meter_std": 0.00012958116712979972, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.999450147151947, "reward_total_composite_std": 0.00012958116712979972} {"timestamp_utc": "2026-04-12T03:41:27Z", "mode": "train", "global_step": 3277, "epoch": 0.1316222838092943, "loss": 0.0383, "grad_norm": 3.1708502769470215, "learning_rate": 7.272727272727274e-08, "num_tokens": 7465634.0, "completions/mean_length": 285.875, "completions/min_length": 275.0, "completions/max_length": 308.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 285.875, "completions/min_terminated_length": 275.0, "completions/max_terminated_length": 308.0, "rewards/meter/mean": 0.9972749948501587, "rewards/meter/std": 0.0005863041151314974, "rewards/count_adherence/mean": 0.9722222089767456, "rewards/count_adherence/std": 0.05143444985151291, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9500774145126343, "rewards/repeat_penalty/std": 0.03746138885617256, "rewards/total_composite/mean": 0.9212331175804138, "rewards/total_composite/std": 0.062395479530096054, "reward": 0.9212331175804138, "reward_std": 0.06239548325538635, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.039440274238586426, "sampling/sampling_logp_difference/max": 1.6334123611450195, "sampling/importance_sampling_ratio/min": 0.1952621191740036, "sampling/importance_sampling_ratio/mean": 1.0104100704193115, "sampling/importance_sampling_ratio/max": 1.906610369682312, "entropy": 0.3647267259657383, "clip_ratio/low_mean": 0.007274207135196775, "clip_ratio/low_min": 0.007274207135196775, "clip_ratio/high_mean": 0.018823198741301894, "clip_ratio/high_max": 0.018823198741301894, "clip_ratio/region_mean": 0.02609740587649867, "reward_total_mean": 0.9212331175804138, "reward_meter_mean": 0.9972749948501587, "reward_meter_std": 0.0005863041151314974, "reward_count_adherence_mean": 0.9722222089767456, "reward_count_adherence_std": 0.05143444985151291, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9500774145126343, "reward_repeat_penalty_std": 0.03746138885617256, "reward_total_composite_mean": 0.9212331175804138, "reward_total_composite_std": 0.062395479530096054} {"timestamp_utc": "2026-04-12T03:41:34Z", "mode": "train", "global_step": 3278, "epoch": 0.13166244929107926, "loss": 0.0041, "grad_norm": 2.639291286468506, "learning_rate": 6.96969696969697e-08, "num_tokens": 7469932.0, "completions/mean_length": 333.25, "completions/min_length": 331.0, "completions/max_length": 337.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 333.25, "completions/min_terminated_length": 331.0, "completions/max_terminated_length": 337.0, "rewards/meter/mean": 0.9980206489562988, "rewards/meter/std": 0.0009543930646032095, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9934210777282715, "rewards/repeat_penalty/std": 0.01860806532204151, "rewards/total_composite/mean": 0.9914476871490479, "rewards/total_composite/std": 0.01817752607166767, "reward": 0.9914476871490479, "reward_std": 0.01817752793431282, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0520353838801384, "sampling/sampling_logp_difference/max": 1.9485530853271484, "sampling/importance_sampling_ratio/min": 0.14248007535934448, "sampling/importance_sampling_ratio/mean": 1.010366678237915, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4740913324058056, "clip_ratio/low_mean": 0.005647590383887291, "clip_ratio/low_min": 0.005647590383887291, "clip_ratio/high_mean": 0.03409638348966837, "clip_ratio/high_max": 0.03409638348966837, "clip_ratio/region_mean": 0.03974397387355566, "reward_total_mean": 0.9914476871490479, "reward_meter_mean": 0.9980206489562988, "reward_meter_std": 0.0009543930646032095, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9934210777282715, "reward_repeat_penalty_std": 0.01860806532204151, "reward_total_composite_mean": 0.9914476871490479, "reward_total_composite_std": 0.01817752607166767} {"timestamp_utc": "2026-04-12T03:41:39Z", "mode": "train", "global_step": 3279, "epoch": 0.1317026147728642, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.666666666666668e-08, "num_tokens": 7471564.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 8.241144678322598e-05, "sampling/sampling_logp_difference/max": 0.0017932088812813163, "sampling/importance_sampling_ratio/min": 0.9982084035873413, "sampling/importance_sampling_ratio/mean": 1.0000587701797485, "sampling/importance_sampling_ratio/max": 1.0012202262878418, "entropy": 0.0007044602243695408, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:41:44Z", "mode": "train", "global_step": 3280, "epoch": 0.13174278025464917, "loss": 0.0002, "grad_norm": 0.038945455104112625, "learning_rate": 6.363636363636365e-08, "num_tokens": 7473684.0, "completions/mean_length": 98.0, "completions/min_length": 98.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.0, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9994074702262878, "rewards/meter/std": 3.4306899578950834e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994074702262878, "rewards/total_composite/std": 3.4306899578950834e-06, "reward": 0.9994074702262878, "reward_std": 3.428247964620823e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0032499232329428196, "sampling/sampling_logp_difference/max": 0.3044092655181885, "sampling/importance_sampling_ratio/min": 0.7375589609146118, "sampling/importance_sampling_ratio/mean": 1.0016450881958008, "sampling/importance_sampling_ratio/max": 1.345357894897461, "entropy": 0.033778895856812596, "clip_ratio/low_mean": 0.006377550889737904, "clip_ratio/low_min": 0.006377550889737904, "clip_ratio/high_mean": 0.0025510203558951616, "clip_ratio/high_max": 0.0025510203558951616, "clip_ratio/region_mean": 0.008928571245633066, "reward_total_mean": 0.9994074702262878, "reward_meter_mean": 0.9994074702262878, "reward_meter_std": 3.4306899578950834e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994074702262878, "reward_total_composite_std": 3.4306899578950834e-06} {"timestamp_utc": "2026-04-12T03:41:48Z", "mode": "train", "global_step": 3281, "epoch": 0.13178294573643412, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.060606060606061e-08, "num_tokens": 7475108.0, "completions/mean_length": 34.0, "completions/min_length": 34.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9977458119392395, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977458119392395, "rewards/total_composite/std": 0.0, "reward": 0.9977458119392395, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.002193058142438531, "sampling/sampling_logp_difference/max": 0.06379712373018265, "sampling/importance_sampling_ratio/min": 0.9427482485771179, "sampling/importance_sampling_ratio/mean": 1.0015556812286377, "sampling/importance_sampling_ratio/max": 1.0658761262893677, "entropy": 0.024989124154672027, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9977458119392395, "reward_meter_mean": 0.9977458119392395, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9977458119392395, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:41:53Z", "mode": "train", "global_step": 3282, "epoch": 0.13182311121821907, "loss": 0.0002, "grad_norm": 0.13846832513809204, "learning_rate": 5.757575757575758e-08, "num_tokens": 7476965.0, "completions/mean_length": 71.125, "completions/min_length": 71.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.125, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9994271993637085, "rewards/meter/std": 1.0313916391169187e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994271993637085, "rewards/total_composite/std": 1.0313916391169187e-05, "reward": 0.9994271993637085, "reward_std": 1.0320004548702855e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008291705511510372, "sampling/sampling_logp_difference/max": 0.5785994529724121, "sampling/importance_sampling_ratio/min": 0.5606830716133118, "sampling/importance_sampling_ratio/mean": 1.0017709732055664, "sampling/importance_sampling_ratio/max": 1.1672364473342896, "entropy": 0.0743112824857235, "clip_ratio/low_mean": 0.0017361111240461469, "clip_ratio/low_min": 0.0017361111240461469, "clip_ratio/high_mean": 0.0017605633474886417, "clip_ratio/high_max": 0.0017605633474886417, "clip_ratio/region_mean": 0.0034966744715347886, "reward_total_mean": 0.9994271993637085, "reward_meter_mean": 0.9994271993637085, "reward_meter_std": 1.0313916391169187e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9994271993637085, "reward_total_composite_std": 1.0313916391169187e-05} {"timestamp_utc": "2026-04-12T03:41:57Z", "mode": "train", "global_step": 3283, "epoch": 0.13186327670000403, "loss": -0.0006, "grad_norm": 0.04218384996056557, "learning_rate": 5.454545454545455e-08, "num_tokens": 7478764.0, "completions/mean_length": 66.875, "completions/min_length": 66.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.875, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9981493353843689, "rewards/meter/std": 8.558939043723512e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981493353843689, "rewards/total_composite/std": 8.558939043723512e-06, "reward": 0.9981493353843689, "reward_std": 8.561799404560588e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005590750835835934, "sampling/sampling_logp_difference/max": 0.4178617000579834, "sampling/importance_sampling_ratio/min": 0.8036051392555237, "sampling/importance_sampling_ratio/mean": 1.002824306488037, "sampling/importance_sampling_ratio/max": 1.5187106132507324, "entropy": 0.03944724937900901, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/high_mean": 0.005597014795057476, "clip_ratio/high_max": 0.005597014795057476, "clip_ratio/region_mean": 0.009384893695823848, "reward_total_mean": 0.9981493353843689, "reward_meter_mean": 0.9981493353843689, "reward_meter_std": 8.558939043723512e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9981493353843689, "reward_total_composite_std": 8.558939043723512e-06} {"timestamp_utc": "2026-04-12T03:42:03Z", "mode": "train", "global_step": 3284, "epoch": 0.13190344218178898, "loss": -0.0012, "grad_norm": 1.1970688104629517, "learning_rate": 5.151515151515152e-08, "num_tokens": 7481306.0, "completions/mean_length": 131.75, "completions/min_length": 131.0, "completions/max_length": 132.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.75, "completions/min_terminated_length": 131.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.9992576837539673, "rewards/meter/std": 0.00048146978951990604, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992576837539673, "rewards/total_composite/std": 0.00048146978951990604, "reward": 0.9992576837539673, "reward_std": 0.0004814636486116797, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009545176289975643, "sampling/sampling_logp_difference/max": 0.7750411033630371, "sampling/importance_sampling_ratio/min": 0.4606848359107971, "sampling/importance_sampling_ratio/mean": 1.0037535429000854, "sampling/importance_sampling_ratio/max": 1.4763288497924805, "entropy": 0.08821968082338572, "clip_ratio/low_mean": 0.0009469697251915932, "clip_ratio/low_min": 0.0009469697251915932, "clip_ratio/high_mean": 0.0019083969527855515, "clip_ratio/high_max": 0.0019083969527855515, "clip_ratio/region_mean": 0.0028553666779771447, "reward_total_mean": 0.9992576837539673, "reward_meter_mean": 0.9992576837539673, "reward_meter_std": 0.00048146978951990604, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9992576837539673, "reward_total_composite_std": 0.00048146978951990604} {"timestamp_utc": "2026-04-12T03:42:11Z", "mode": "train", "global_step": 3285, "epoch": 0.13194360766357394, "loss": -0.0227, "grad_norm": 1.6411669254302979, "learning_rate": 4.8484848484848486e-08, "num_tokens": 7485883.0, "completions/mean_length": 361.125, "completions/min_length": 347.0, "completions/max_length": 385.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 361.125, "completions/min_terminated_length": 347.0, "completions/max_terminated_length": 385.0, "rewards/meter/mean": 0.9987712502479553, "rewards/meter/std": 0.0013375915586948395, "rewards/count_adherence/mean": 0.8409091234207153, "rewards/count_adherence/std": 0.04208274558186531, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9852941036224365, "rewards/repeat_penalty/std": 0.027230001986026764, "rewards/total_composite/mean": 0.827822208404541, "rewards/total_composite/std": 0.05304892361164093, "reward": 0.827822208404541, "reward_std": 0.053048938512802124, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04496872425079346, "sampling/sampling_logp_difference/max": 1.6645660400390625, "sampling/importance_sampling_ratio/min": 0.18927277624607086, "sampling/importance_sampling_ratio/mean": 1.01465904712677, "sampling/importance_sampling_ratio/max": 1.6690725088119507, "entropy": 0.4625861681997776, "clip_ratio/low_mean": 0.01729804463684559, "clip_ratio/low_min": 0.01729804463684559, "clip_ratio/high_mean": 0.005871425149962306, "clip_ratio/high_max": 0.005871425149962306, "clip_ratio/region_mean": 0.023169469786807895, "reward_total_mean": 0.827822208404541, "reward_meter_mean": 0.9987712502479553, "reward_meter_std": 0.0013375915586948395, "reward_count_adherence_mean": 0.8409091234207153, "reward_count_adherence_std": 0.04208274558186531, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9852941036224365, "reward_repeat_penalty_std": 0.027230001986026764, "reward_total_composite_mean": 0.827822208404541, "reward_total_composite_std": 0.05304892361164093} {"timestamp_utc": "2026-04-12T03:42:18Z", "mode": "train", "global_step": 3286, "epoch": 0.1319837731453589, "loss": 0.0016, "grad_norm": 0.6697944402694702, "learning_rate": 4.545454545454546e-08, "num_tokens": 7488852.0, "completions/mean_length": 178.125, "completions/min_length": 177.0, "completions/max_length": 180.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 178.125, "completions/min_terminated_length": 177.0, "completions/max_terminated_length": 180.0, "rewards/meter/mean": 0.9990102052688599, "rewards/meter/std": 0.00010323245805921033, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990102052688599, "rewards/total_composite/std": 0.00010323245805921033, "reward": 0.9990102052688599, "reward_std": 0.0001032423970173113, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02063995786011219, "sampling/sampling_logp_difference/max": 0.810206413269043, "sampling/importance_sampling_ratio/min": 0.44476625323295593, "sampling/importance_sampling_ratio/mean": 1.006458044052124, "sampling/importance_sampling_ratio/max": 1.7193830013275146, "entropy": 0.24275393225252628, "clip_ratio/low_mean": 0.0021067415946163237, "clip_ratio/low_min": 0.0021067415946163237, "clip_ratio/high_mean": 0.008415504998993129, "clip_ratio/high_max": 0.008415504998993129, "clip_ratio/region_mean": 0.010522246593609452, "reward_total_mean": 0.9990102052688599, "reward_meter_mean": 0.9990102052688599, "reward_meter_std": 0.00010323245805921033, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9990102052688599, "reward_total_composite_std": 0.00010323245805921033} {"timestamp_utc": "2026-04-12T03:42:23Z", "mode": "train", "global_step": 3287, "epoch": 0.13202393862714384, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 4.2424242424242435e-08, "num_tokens": 7490484.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002587019116617739, "sampling/sampling_logp_difference/max": 0.0030488455668091774, "sampling/importance_sampling_ratio/min": 0.9995039105415344, "sampling/importance_sampling_ratio/mean": 1.0002548694610596, "sampling/importance_sampling_ratio/max": 1.0030535459518433, "entropy": 0.0020569458283716813, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:42:29Z", "mode": "train", "global_step": 3288, "epoch": 0.1320641041089288, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 3.93939393939394e-08, "num_tokens": 7492204.0, "completions/mean_length": 64.0, "completions/min_length": 64.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "reward": 0.9993994235992432, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 9.810684423428029e-05, "sampling/sampling_logp_difference/max": 0.0028118479531258345, "sampling/importance_sampling_ratio/min": 0.9984528422355652, "sampling/importance_sampling_ratio/mean": 1.0000782012939453, "sampling/importance_sampling_ratio/max": 1.002815842628479, "entropy": 0.001106615709431935, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9993994235992432, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:42:34Z", "mode": "train", "global_step": 3289, "epoch": 0.13210426959071375, "loss": 0.0109, "grad_norm": 3.059532642364502, "learning_rate": 3.636363636363637e-08, "num_tokens": 7494281.0, "completions/mean_length": 103.625, "completions/min_length": 101.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.625, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.9926178455352783, "rewards/meter/std": 0.0021126074716448784, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9677305221557617, "rewards/total_composite/std": 0.06925326585769653, "reward": 0.9677305221557617, "reward_std": 0.06925328075885773, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03573079779744148, "sampling/sampling_logp_difference/max": 1.4160699844360352, "sampling/importance_sampling_ratio/min": 0.2426658272743225, "sampling/importance_sampling_ratio/mean": 1.0063260793685913, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2910528890788555, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.025478466413915157, "clip_ratio/high_max": 0.025478466413915157, "clip_ratio/region_mean": 0.025478466413915157, "reward_total_mean": 0.9677305221557617, "reward_meter_mean": 0.9926178455352783, "reward_meter_std": 0.0021126074716448784, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.9677305221557617, "reward_total_composite_std": 0.06925326585769653} {"timestamp_utc": "2026-04-12T03:42:40Z", "mode": "train", "global_step": 3290, "epoch": 0.1321444350724987, "loss": 0.0003, "grad_norm": 2.9757983684539795, "learning_rate": 3.333333333333334e-08, "num_tokens": 7497469.0, "completions/mean_length": 203.5, "completions/min_length": 201.0, "completions/max_length": 209.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 203.5, "completions/min_terminated_length": 201.0, "completions/max_terminated_length": 209.0, "rewards/meter/mean": 0.9778822660446167, "rewards/meter/std": 0.03634575754404068, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9772727489471436, "rewards/repeat_penalty/std": 0.04208271950483322, "rewards/total_composite/mean": 0.9552866816520691, "rewards/total_composite/std": 0.0472070537507534, "reward": 0.9552866816520691, "reward_std": 0.047207050025463104, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.038450971245765686, "sampling/sampling_logp_difference/max": 1.2187190055847168, "sampling/importance_sampling_ratio/min": 0.29560860991477966, "sampling/importance_sampling_ratio/mean": 1.0083707571029663, "sampling/importance_sampling_ratio/max": 1.8156616687774658, "entropy": 0.38689322397112846, "clip_ratio/low_mean": 0.009898273507133126, "clip_ratio/low_min": 0.009898273507133126, "clip_ratio/high_mean": 0.025775008834898472, "clip_ratio/high_max": 0.025775008834898472, "clip_ratio/region_mean": 0.0356732823420316, "reward_total_mean": 0.9552866816520691, "reward_meter_mean": 0.9778822660446167, "reward_meter_std": 0.03634575754404068, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9772727489471436, "reward_repeat_penalty_std": 0.04208271950483322, "reward_total_composite_mean": 0.9552866816520691, "reward_total_composite_std": 0.0472070537507534} {"timestamp_utc": "2026-04-12T03:42:45Z", "mode": "train", "global_step": 3291, "epoch": 0.13218460055428366, "loss": 0.0059, "grad_norm": 2.9630963802337646, "learning_rate": 3.0303030303030305e-08, "num_tokens": 7499282.0, "completions/mean_length": 65.625, "completions/min_length": 65.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.625, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9937740564346313, "rewards/meter/std": 0.0003091543912887573, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9937740564346313, "rewards/total_composite/std": 0.0003091543912887573, "reward": 0.9937740564346313, "reward_std": 0.0003091563412453979, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01210538949817419, "sampling/sampling_logp_difference/max": 0.9835647940635681, "sampling/importance_sampling_ratio/min": 0.37397557497024536, "sampling/importance_sampling_ratio/mean": 1.0025181770324707, "sampling/importance_sampling_ratio/max": 1.6661887168884277, "entropy": 0.06898282328620553, "clip_ratio/low_mean": 0.005710955825634301, "clip_ratio/low_min": 0.005710955825634301, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.005710955825634301, "reward_total_mean": 0.9937740564346313, "reward_meter_mean": 0.9937740564346313, "reward_meter_std": 0.0003091543912887573, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9937740564346313, "reward_total_composite_std": 0.0003091543912887573} {"timestamp_utc": "2026-04-12T03:42:49Z", "mode": "train", "global_step": 3292, "epoch": 0.13222476603606861, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 2.7272727272727276e-08, "num_tokens": 7501162.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.000192559469724074, "sampling/sampling_logp_difference/max": 0.0017719701863825321, "sampling/importance_sampling_ratio/min": 0.9987837076187134, "sampling/importance_sampling_ratio/mean": 1.000186800956726, "sampling/importance_sampling_ratio/max": 1.001773476600647, "entropy": 0.0016719198902137578, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:42:59Z", "mode": "train", "global_step": 3293, "epoch": 0.13226493151785357, "loss": -0.0169, "grad_norm": 1.2802501916885376, "learning_rate": 2.4242424242424243e-08, "num_tokens": 7506578.0, "completions/mean_length": 428.0, "completions/min_length": 408.0, "completions/max_length": 448.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 428.0, "completions/min_terminated_length": 408.0, "completions/max_terminated_length": 448.0, "rewards/meter/mean": 0.9988534450531006, "rewards/meter/std": 0.00014253715926315635, "rewards/count_adherence/mean": 0.8392857313156128, "rewards/count_adherence/std": 0.03306501731276512, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9562541246414185, "rewards/repeat_penalty/std": 0.03288932517170906, "rewards/total_composite/mean": 0.8016811609268188, "rewards/total_composite/std": 0.04294924437999725, "reward": 0.8016811609268188, "reward_std": 0.042949263006448746, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03670765459537506, "sampling/sampling_logp_difference/max": 1.6504631042480469, "sampling/importance_sampling_ratio/min": 0.19196099042892456, "sampling/importance_sampling_ratio/mean": 1.0091326236724854, "sampling/importance_sampling_ratio/max": 1.804500699043274, "entropy": 0.39450087025761604, "clip_ratio/low_mean": 0.008654525852762163, "clip_ratio/low_min": 0.008654525852762163, "clip_ratio/high_mean": 0.014237138908356428, "clip_ratio/high_max": 0.014237138908356428, "clip_ratio/region_mean": 0.02289166476111859, "reward_total_mean": 0.8016811609268188, "reward_meter_mean": 0.9988534450531006, "reward_meter_std": 0.00014253715926315635, "reward_count_adherence_mean": 0.8392857313156128, "reward_count_adherence_std": 0.03306501731276512, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9562541246414185, "reward_repeat_penalty_std": 0.03288932517170906, "reward_total_composite_mean": 0.8016811609268188, "reward_total_composite_std": 0.04294924437999725} {"timestamp_utc": "2026-04-12T03:43:03Z", "mode": "train", "global_step": 3294, "epoch": 0.13230509699963852, "loss": -0.0, "grad_norm": 0.030057502910494804, "learning_rate": 2.1212121212121217e-08, "num_tokens": 7508354.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9973385334014893, "rewards/meter/std": 1.0115243185282452e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973385334014893, "rewards/total_composite/std": 1.0115243185282452e-06, "reward": 0.9973385334014893, "reward_std": 1.0115243185282452e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.002317759208381176, "sampling/sampling_logp_difference/max": 0.680436372756958, "sampling/importance_sampling_ratio/min": 0.5063959360122681, "sampling/importance_sampling_ratio/mean": 0.9997027516365051, "sampling/importance_sampling_ratio/max": 1.0437626838684082, "entropy": 0.012461528764106333, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9973385334014893, "reward_meter_mean": 0.9973385334014893, "reward_meter_std": 1.0115243185282452e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9973385334014893, "reward_total_composite_std": 1.0115243185282452e-06} {"timestamp_utc": "2026-04-12T03:43:08Z", "mode": "train", "global_step": 3295, "epoch": 0.13234526248142348, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 1.8181818181818185e-08, "num_tokens": 7510050.0, "completions/mean_length": 54.0, "completions/min_length": 54.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "reward": 0.787638783454895, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00023673544637858868, "sampling/sampling_logp_difference/max": 0.002077840268611908, "sampling/importance_sampling_ratio/min": 0.9994556307792664, "sampling/importance_sampling_ratio/mean": 1.000232458114624, "sampling/importance_sampling_ratio/max": 1.002079963684082, "entropy": 0.0021129547676537186, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.787638783454895, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-12T03:43:14Z", "mode": "train", "global_step": 3296, "epoch": 0.13238542796320843, "loss": 0.0072, "grad_norm": 1.9730912446975708, "learning_rate": 1.5151515151515152e-08, "num_tokens": 7513219.0, "completions/mean_length": 197.125, "completions/min_length": 192.0, "completions/max_length": 200.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 197.125, "completions/min_terminated_length": 192.0, "completions/max_terminated_length": 200.0, "rewards/meter/mean": 0.9992784261703491, "rewards/meter/std": 0.00010585635027382523, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9886363744735718, "rewards/repeat_penalty/std": 0.03214120864868164, "rewards/total_composite/mean": 0.9879227876663208, "rewards/total_composite/std": 0.0321127250790596, "reward": 0.9879227876663208, "reward_std": 0.032112736254930496, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021909546107053757, "sampling/sampling_logp_difference/max": 0.8929929733276367, "sampling/importance_sampling_ratio/min": 0.40942853689193726, "sampling/importance_sampling_ratio/mean": 1.009697675704956, "sampling/importance_sampling_ratio/max": 1.8454686403274536, "entropy": 0.2336829099804163, "clip_ratio/low_mean": 0.0025125627871602774, "clip_ratio/low_min": 0.0025125627871602774, "clip_ratio/high_mean": 0.012068168085534126, "clip_ratio/high_max": 0.012068168085534126, "clip_ratio/region_mean": 0.014580730872694403, "reward_total_mean": 0.9879227876663208, "reward_meter_mean": 0.9992784261703491, "reward_meter_std": 0.00010585635027382523, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9886363744735718, "reward_repeat_penalty_std": 0.03214120864868164, "reward_total_composite_mean": 0.9879227876663208, "reward_total_composite_std": 0.0321127250790596} {"timestamp_utc": "2026-04-12T03:43:22Z", "mode": "train", "global_step": 3297, "epoch": 0.13242559344499338, "loss": -0.0017, "grad_norm": 2.3195743560791016, "learning_rate": 1.2121212121212122e-08, "num_tokens": 7517109.0, "completions/mean_length": 302.25, "completions/min_length": 291.0, "completions/max_length": 316.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 302.25, "completions/min_terminated_length": 291.0, "completions/max_terminated_length": 316.0, "rewards/meter/mean": 0.9957056641578674, "rewards/meter/std": 0.006368847563862801, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9926470518112183, "rewards/repeat_penalty/std": 0.020797256380319595, "rewards/total_composite/mean": 0.9883631467819214, "rewards/total_composite/std": 0.020583728328347206, "reward": 0.9883631467819214, "reward_std": 0.0205837395042181, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05743308737874031, "sampling/sampling_logp_difference/max": 1.1259183883666992, "sampling/importance_sampling_ratio/min": 0.3243544399738312, "sampling/importance_sampling_ratio/mean": 1.014825701713562, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5184807442128658, "clip_ratio/low_mean": 0.010391832329332829, "clip_ratio/low_min": 0.010391832329332829, "clip_ratio/high_mean": 0.03054804727435112, "clip_ratio/high_max": 0.03054804727435112, "clip_ratio/region_mean": 0.04093987960368395, "reward_total_mean": 0.9883631467819214, "reward_meter_mean": 0.9957056641578674, "reward_meter_std": 0.006368847563862801, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9926470518112183, "reward_repeat_penalty_std": 0.020797256380319595, "reward_total_composite_mean": 0.9883631467819214, "reward_total_composite_std": 0.020583728328347206} {"timestamp_utc": "2026-04-12T03:43:26Z", "mode": "train", "global_step": 3298, "epoch": 0.13246575892677834, "loss": 0.0017, "grad_norm": 2.183478593826294, "learning_rate": 9.090909090909092e-09, "num_tokens": 7518824.0, "completions/mean_length": 65.375, "completions/min_length": 65.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.375, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9922501444816589, "rewards/meter/std": 0.004244488663971424, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9922501444816589, "rewards/total_composite/std": 0.004244488663971424, "reward": 0.9922501444816589, "reward_std": 0.0042444937862455845, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013568882830440998, "sampling/sampling_logp_difference/max": 0.7026901245117188, "sampling/importance_sampling_ratio/min": 0.49525120854377747, "sampling/importance_sampling_ratio/mean": 1.0067819356918335, "sampling/importance_sampling_ratio/max": 1.4506497383117676, "entropy": 0.11512181628495455, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/high_mean": 0.005710955825634301, "clip_ratio/high_max": 0.005710955825634301, "clip_ratio/region_mean": 0.009498834726400673, "reward_total_mean": 0.9922501444816589, "reward_meter_mean": 0.9922501444816589, "reward_meter_std": 0.004244488663971424, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9922501444816589, "reward_total_composite_std": 0.004244488663971424} {"timestamp_utc": "2026-04-12T03:43:32Z", "mode": "train", "global_step": 3299, "epoch": 0.1325059244085633, "loss": 0.0057, "grad_norm": 3.7170727252960205, "learning_rate": 6.060606060606061e-09, "num_tokens": 7521612.0, "completions/mean_length": 166.5, "completions/min_length": 159.0, "completions/max_length": 170.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 166.5, "completions/min_terminated_length": 159.0, "completions/max_terminated_length": 170.0, "rewards/meter/mean": 0.9987373352050781, "rewards/meter/std": 0.000354176911059767, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9987373352050781, "rewards/total_composite/std": 0.000354176911059767, "reward": 0.9987373352050781, "reward_std": 0.00035417175968177617, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03916100785136223, "sampling/sampling_logp_difference/max": 1.9973111152648926, "sampling/importance_sampling_ratio/min": 0.13569968938827515, "sampling/importance_sampling_ratio/mean": 1.001840591430664, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3029062431305647, "clip_ratio/low_mean": 0.01500677247531712, "clip_ratio/low_min": 0.01500677247531712, "clip_ratio/high_mean": 0.015017259865999222, "clip_ratio/high_max": 0.015017259865999222, "clip_ratio/region_mean": 0.030024032341316342, "reward_total_mean": 0.9987373352050781, "reward_meter_mean": 0.9987373352050781, "reward_meter_std": 0.000354176911059767, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9987373352050781, "reward_total_composite_std": 0.000354176911059767} {"timestamp_utc": "2026-04-12T03:43:36Z", "mode": "train", "global_step": 3300, "epoch": 0.13254608989034825, "loss": 0.0153, "grad_norm": 6.677674293518066, "learning_rate": 3.0303030303030304e-09, "num_tokens": 7523188.0, "completions/mean_length": 46.0, "completions/min_length": 46.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.0, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.9442145824432373, "rewards/meter/std": 0.003297017654404044, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9442145824432373, "rewards/total_composite/std": 0.003297017654404044, "reward": 0.9442145824432373, "reward_std": 0.0032970213796943426, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021698685362935066, "sampling/sampling_logp_difference/max": 2.1988744735717773, "sampling/importance_sampling_ratio/min": 0.11092793941497803, "sampling/importance_sampling_ratio/mean": 1.0027215480804443, "sampling/importance_sampling_ratio/max": 1.663463830947876, "entropy": 0.09115941543132067, "clip_ratio/low_mean": 0.00815217406488955, "clip_ratio/low_min": 0.00815217406488955, "clip_ratio/high_mean": 0.0027173913549631834, "clip_ratio/high_max": 0.0027173913549631834, "clip_ratio/region_mean": 0.010869565419852734, "reward_total_mean": 0.9442145824432373, "reward_meter_mean": 0.9442145824432373, "reward_meter_std": 0.003297017654404044, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9442145824432373, "reward_total_composite_std": 0.003297017654404044} {"timestamp_utc": "2026-04-12T03:44:57Z", "mode": "eval", "global_step": 3300, "epoch": 0.13254608989034825, "eval_loss": NaN, "eval_runtime": 80.3156, "eval_samples_per_second": 1.295, "eval_steps_per_second": 0.162, "eval_num_tokens": 7523188.0, "eval_completions/mean_length": 213.14423076923077, "eval_completions/min_length": 61.23076923076923, "eval_completions/max_length": 421.38461538461536, "eval_completions/clipped_ratio": 0.009615384615384616, "eval_completions/mean_terminated_length": 209.98489027756912, "eval_completions/min_terminated_length": 61.23076923076923, "eval_completions/max_terminated_length": 412.2307692307692, "eval_rewards/meter/mean": 0.7804106657321637, "eval_rewards/meter/std": 0.3429693900621854, "eval_rewards/count_adherence/mean": 0.9621203220807589, "eval_rewards/count_adherence/std": 0.06154714152216911, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/repeat_penalty/mean": 0.9574309633328364, "eval_rewards/repeat_penalty/std": 0.07397974411455485, "eval_rewards/total_composite/mean": 0.7239617155148432, "eval_rewards/total_composite/std": 0.3431667788670613, "eval_reward": 0.7239617155148432, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.03249364231641476, "eval_sampling/sampling_logp_difference/max": 1.2331663278432994, "eval_sampling/importance_sampling_ratio/min": 0.2969068231490942, "eval_sampling/importance_sampling_ratio/mean": 1.0095902956449068, "eval_sampling/importance_sampling_ratio/max": 1.6007587084403405, "eval_entropy": 0.38235602699793303, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.7239617155148432, "eval_reward_meter_mean": 0.7804106657321637, "eval_reward_meter_std": 0.3429693900621854, "eval_reward_count_adherence_mean": 0.9621203220807589, "eval_reward_count_adherence_std": 0.06154714152216911, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_repeat_penalty_mean": 0.9574309633328364, "eval_reward_repeat_penalty_std": 0.07397974411455485, "eval_reward_total_composite_mean": 0.7239617155148432, "eval_reward_total_composite_std": 0.3431667788670613} {"timestamp_utc": "2026-04-12T03:45:00Z", "mode": "train", "global_step": 3300, "epoch": 0.13254608989034825, "train_runtime": 18489.879, "train_samples_per_second": 1.428, "train_steps_per_second": 0.178, "total_flos": 0.0, "train_loss": 0.0004697396774788627}